[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"glossary-rag-chunking::en":3,"gloss-cluster-rag-chunking::en":23,"gloss-next-rag-chunking::en":9},{"slug":4,"category":5,"name":6,"definition":7,"meta_desc":8,"faq":9,"schema_markup":9,"related":10},"rag-chunking","data-infra","RAG Chunking","Chunking is the step in a retrieval pipeline where documents are split into passages before being embedded and indexed. It sounds like plumbing and it is the single biggest determinant of retrieval quality in most RAG systems.\n\nThe trade-off runs in both directions. Small chunks embed precisely and retrieve cleanly, but they arrive at the model stripped of the surrounding context that made them meaningful — a paragraph that says \"this does not apply to enterprise plans\" is dangerous without the paragraph above it. Large chunks keep context and dilute the embedding, so a page about twelve topics matches weakly on all of them and strongly on none.\n\nWhat works in practice is chunking on structure rather than character count: headings, list items, table rows, function definitions. Overlap of a sentence or two carries context across boundaries, and storing the parent document's title and section path with each chunk lets you re-expand at answer time.\n\nMeasure it. Retrieval precision on a fixed question set is cheap to compute and will tell you more about your assistant's accuracy than swapping the model will.","Chunking splits documents before embedding them. Chunk too small and answers lose context; too large and retrieval returns noise.",null,[11,14,17,20],{"slug":12,"name":13},"context-window","Context Window",{"slug":15,"name":16},"embedding","Embedding",{"slug":18,"name":19},"retrieval-augmented-generation","Retrieval-Augmented Generation (RAG)",{"slug":21,"name":22},"vector-database","Vector Database",[24,28,31,34,37,41,44,47,50,53,57,60],{"slug":25,"category":5,"name":26,"updated_at":27},"acid","ACID","2026-08-24T02:46:37+00:00",{"slug":29,"category":5,"name":30,"updated_at":27},"ann-search","ANN Search",{"slug":32,"category":5,"name":33,"updated_at":27},"backpressure","Backpressure",{"slug":35,"category":5,"name":36,"updated_at":27},"batch-processing","Batch Processing",{"slug":38,"category":5,"name":39,"updated_at":40},"bm25","BM25","2026-08-24T02:46:38+00:00",{"slug":42,"category":5,"name":43,"updated_at":27},"cache","Cache",{"slug":45,"category":5,"name":46,"updated_at":27},"cap-theorem","CAP Theorem",{"slug":48,"category":5,"name":49,"updated_at":27},"change-data-capture","Change Data Capture (CDC)",{"slug":51,"category":5,"name":52,"updated_at":27},"chroma","Chroma",{"slug":54,"category":5,"name":55,"updated_at":56},"chunk-overlap","Chunk Overlap","2026-08-24T03:30:02+00:00",{"slug":58,"category":5,"name":59,"updated_at":27},"columnar-storage","Columnar Storage",{"slug":61,"category":5,"name":62,"updated_at":27},"connection-pooling","Connection Pooling"]