[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"glossary-lakehouse::en":3,"gloss-cluster-lakehouse::en":23,"gloss-next-lakehouse::en":9},{"slug":4,"category":5,"name":6,"definition":7,"meta_desc":8,"faq":9,"schema_markup":9,"related":10},"lakehouse","data-infra","Data Lakehouse","A lakehouse is an architecture that puts data-warehouse features — ACID transactions, schema enforcement, fast SQL — directly on top of cheap object storage like S3, rather than in a proprietary warehouse. It aims to combine the low cost and flexibility of a data lake with the reliability and performance of a warehouse, so you don't maintain two separate systems.\n\nThe enabling technology is open table formats — Apache Iceberg, Delta Lake, and Apache Hudi — which add a transaction log and metadata layer over columnar Parquet files. That layer brings atomic writes, time travel, and schema evolution to files that would otherwise be a messy \"data swamp\".\n\nFor SaaS builders, the lakehouse pitch is avoiding lock-in and duplicate storage: your raw data and your analytics-ready tables live in the same open files, queryable by many engines (Spark, Trino, DuckDB, Snowflake, Databricks). Practical note: the promise is real but the tooling is younger than classic warehouses. If your data is modest, a managed warehouse is often simpler; reach for a lakehouse when scale, cost, or multi-engine access justify the extra moving parts.","A lakehouse puts warehouse features — ACID transactions, schema enforcement, fast SQL — directly on cheap object storage like S3 instead of a proprietary warehouse.",null,[11,14,17,20],{"slug":12,"name":13},"columnar-storage","Columnar Storage",{"slug":15,"name":16},"data-lake","Data Lake",{"slug":18,"name":19},"data-warehouse","Data Warehouse",{"slug":21,"name":22},"olap","OLAP (Online Analytical Processing)",[24,28,31,34,37,41,44,47,50,53,57,58],{"slug":25,"category":5,"name":26,"updated_at":27},"acid","ACID","2026-08-24T02:46:37+00:00",{"slug":29,"category":5,"name":30,"updated_at":27},"ann-search","ANN Search",{"slug":32,"category":5,"name":33,"updated_at":27},"backpressure","Backpressure",{"slug":35,"category":5,"name":36,"updated_at":27},"batch-processing","Batch Processing",{"slug":38,"category":5,"name":39,"updated_at":40},"bm25","BM25","2026-08-24T02:46:38+00:00",{"slug":42,"category":5,"name":43,"updated_at":27},"cache","Cache",{"slug":45,"category":5,"name":46,"updated_at":27},"cap-theorem","CAP Theorem",{"slug":48,"category":5,"name":49,"updated_at":27},"change-data-capture","Change Data Capture (CDC)",{"slug":51,"category":5,"name":52,"updated_at":27},"chroma","Chroma",{"slug":54,"category":5,"name":55,"updated_at":56},"chunk-overlap","Chunk Overlap","2026-08-24T03:30:02+00:00",{"slug":12,"category":5,"name":13,"updated_at":27},{"slug":59,"category":5,"name":60,"updated_at":27},"connection-pooling","Connection Pooling"]