[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"glossary-golden-dataset::en":3,"gloss-cluster-golden-dataset::en":26,"gloss-next-golden-dataset::en":9},{"slug":4,"category":5,"name":6,"definition":7,"meta_desc":8,"faq":9,"schema_markup":9,"related":10},"golden-dataset","prompt-eng","Golden Dataset","A golden dataset is a curated, version-controlled set of representative inputs paired with known-good expected outputs (or acceptance criteria), used to test an AI feature the way unit tests guard normal code. It's the missing piece that turns 'the demo looked fine' into an actual regression check: before you change a prompt, swap a model, or upgrade a provider, you run the new configuration against the golden set and compare scores. Cases should mirror production — including the messy edge cases, adversarial inputs, and known failure modes you've hit — not just happy-path examples. For builders, this is the single highest-leverage habit for shipping reliable AI: start with 20–50 hand-labeled cases, grow it by promoting real failures into permanent test cases, and score with exact match, assertions, or an LLM judge. Caveats: keep it out of any training or few-shot context to avoid contamination, review labels periodically (your definition of 'good' drifts), and remember a small biased set gives false confidence.","A golden dataset is a version-controlled set of representative inputs with known-good outputs — what turns \"the demo looked fine\" into a real regression check.",null,[11,14,17,20,23],{"slug":12,"name":13},"benchmark-contamination","Benchmark Contamination",{"slug":15,"name":16},"llm-as-judge","LLM-as-Judge",{"slug":18,"name":19},"llm-benchmark","LLM Benchmark",{"slug":21,"name":22},"prompt-testing","Prompt Testing",{"slug":24,"name":25},"regression-testing","Regression Testing",[27,31,34,37,41,44,47,50,53,56,59,62],{"slug":28,"category":5,"name":29,"updated_at":30},"analogical-prompting","Analogical Prompting","2026-08-24T02:46:37+00:00",{"slug":32,"category":5,"name":33,"updated_at":30},"automatic-prompt-optimization","Automatic Prompt Optimization",{"slug":35,"category":5,"name":36,"updated_at":30},"chain-of-density","Chain of Density (CoD)",{"slug":38,"category":5,"name":39,"updated_at":40},"chain-of-thought-prompting","Chain-of-Thought Prompting","2026-08-24T02:46:36+00:00",{"slug":42,"category":5,"name":43,"updated_at":30},"chain-of-verification","Chain-of-Verification",{"slug":45,"category":5,"name":46,"updated_at":40},"chunking","Chunking",{"slug":48,"category":5,"name":49,"updated_at":40},"constrained-decoding","Constrained Decoding",{"slug":51,"category":5,"name":52,"updated_at":40},"context-stuffing","Context Stuffing",{"slug":54,"category":5,"name":55,"updated_at":40},"delimiter","Delimiter",{"slug":57,"category":5,"name":58,"updated_at":30},"directional-stimulus-prompting","Directional Stimulus Prompting",{"slug":60,"category":5,"name":61,"updated_at":30},"emotion-prompting","Emotion Prompting",{"slug":63,"category":5,"name":64,"updated_at":40},"few-shot-prompting","Few-Shot Prompting"]