[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"glossary-pairwise-evaluation::en":3,"gloss-cluster-pairwise-evaluation::en":26,"gloss-next-pairwise-evaluation::en":9},{"slug":4,"category":5,"name":6,"definition":7,"meta_desc":8,"faq":9,"schema_markup":9,"related":10},"pairwise-evaluation","prompt-eng","Pairwise Evaluation","Pairwise evaluation scores model outputs by comparing two of them head-to-head and asking which is better, instead of assigning each an absolute grade. It leans on a well-known fact about both humans and LLM judges: relative judgments ('A or B?') are far more consistent than absolute ones ('rate this 1–10'), where scores drift and cluster. Chatbot Arena popularized the pattern at scale — collecting pairwise votes between anonymous models and converting them into Elo-style rankings. For builders, it's the reliable way to answer 'did my new prompt or model actually get better?': show a judge (or a human) the old and new outputs for each case in your eval set and count wins, ties, and losses. Two things to control: position bias — judges tend to favor whichever answer comes first, so swap the order and average — and the many comparisons needed to rank more than two options. Report win-rate with ties, not a single fragile average score.","Pairwise evaluation asks which of two outputs is better instead of scoring each absolutely — relative judgments are far more consistent for humans and LLM judges.",null,[11,14,17,20,23],{"slug":12,"name":13},"golden-dataset","Golden Dataset",{"slug":15,"name":16},"llm-as-judge","LLM-as-Judge",{"slug":18,"name":19},"position-bias","Position Bias",{"slug":21,"name":22},"prompt-testing","Prompt Testing",{"slug":24,"name":25},"rubric-prompting","Rubric Prompting",[27,31,34,37,41,44,47,50,53,56,59,62],{"slug":28,"category":5,"name":29,"updated_at":30},"analogical-prompting","Analogical Prompting","2026-08-24T02:46:37+00:00",{"slug":32,"category":5,"name":33,"updated_at":30},"automatic-prompt-optimization","Automatic Prompt Optimization",{"slug":35,"category":5,"name":36,"updated_at":30},"chain-of-density","Chain of Density (CoD)",{"slug":38,"category":5,"name":39,"updated_at":40},"chain-of-thought-prompting","Chain-of-Thought Prompting","2026-08-24T02:46:36+00:00",{"slug":42,"category":5,"name":43,"updated_at":30},"chain-of-verification","Chain-of-Verification",{"slug":45,"category":5,"name":46,"updated_at":40},"chunking","Chunking",{"slug":48,"category":5,"name":49,"updated_at":40},"constrained-decoding","Constrained Decoding",{"slug":51,"category":5,"name":52,"updated_at":40},"context-stuffing","Context Stuffing",{"slug":54,"category":5,"name":55,"updated_at":40},"delimiter","Delimiter",{"slug":57,"category":5,"name":58,"updated_at":30},"directional-stimulus-prompting","Directional Stimulus Prompting",{"slug":60,"category":5,"name":61,"updated_at":30},"emotion-prompting","Emotion Prompting",{"slug":63,"category":5,"name":64,"updated_at":40},"few-shot-prompting","Few-Shot Prompting"]