[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"glossary-model-calibration::en":3,"gloss-cluster-model-calibration::en":23,"gloss-next-model-calibration::en":9},{"slug":4,"category":5,"name":6,"definition":7,"meta_desc":8,"faq":9,"schema_markup":9,"related":10},"model-calibration","core-ai","Model Calibration","Calibration describes how well a model's confidence matches its actual accuracy. A well-calibrated model that says it's 80% sure is right about 80% of the time; a poorly calibrated one is confidently wrong or needlessly hesitant. This matters the moment you use a model's confidence to make decisions — auto-approving high-confidence answers, routing low-confidence ones to a human, or thresholding a classifier. If the confidence isn't calibrated, those thresholds are meaningless. Research has found that alignment training can degrade a base model's calibration, so a fluent, assured tone is not evidence of correctness. For builders, don't treat a model sounding sure as being sure. Practical note: if you rely on confidence, measure calibration on your own labeled examples — bucket predictions by stated or implied confidence and check the real hit rate in each bucket. Where you can, derive confidence from token probabilities (logprobs) rather than from the model's self-reported \"I'm 90% sure,\" which is often unreliable.","Calibration is how well a model's stated confidence matches its real accuracy — an 80%-sure well-calibrated model is right about 80% of the time.",null,[11,14,17,20],{"slug":12,"name":13},"hallucination","Hallucination",{"slug":15,"name":16},"llm-benchmark","LLM Benchmark",{"slug":18,"name":19},"logprobs","Logprobs (Log Probabilities)",{"slug":21,"name":22},"temperature","Temperature",[24,28,32,36,39,42,45,48,51,54,57,60],{"slug":25,"category":5,"name":26,"updated_at":27},"agentic","Agentic AI","2026-08-24T02:46:36+00:00",{"slug":29,"category":5,"name":30,"updated_at":31},"alignment-tax","Alignment Tax","2026-08-24T02:46:37+00:00",{"slug":33,"category":5,"name":34,"updated_at":35},"artificial-intelligence","Artificial Intelligence (AI)","2026-08-24T02:46:38+00:00",{"slug":37,"category":5,"name":38,"updated_at":27},"attention","Attention",{"slug":40,"category":5,"name":41,"updated_at":35},"beam-search","Beam Search",{"slug":43,"category":5,"name":44,"updated_at":31},"benchmark-contamination","Benchmark Contamination",{"slug":46,"category":5,"name":47,"updated_at":31},"catastrophic-forgetting","Catastrophic Forgetting",{"slug":49,"category":5,"name":50,"updated_at":35},"computer-vision","Computer Vision",{"slug":52,"category":5,"name":53,"updated_at":31},"constitutional-ai","Constitutional AI",{"slug":55,"category":5,"name":56,"updated_at":27},"context-window","Context Window",{"slug":58,"category":5,"name":59,"updated_at":35},"deep-learning","Deep Learning",{"slug":61,"category":5,"name":62,"updated_at":27},"diffusion-model","Diffusion Model"]