[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"glossary-alignment-tax::en":3,"gloss-cluster-alignment-tax::en":26,"gloss-next-alignment-tax::en":9},{"slug":4,"category":5,"name":6,"definition":7,"meta_desc":8,"faq":9,"schema_markup":9,"related":10},"alignment-tax","core-ai","Alignment Tax","The alignment tax is the capability you give up to make a model safer and better-behaved. Training a model to refuse harmful requests, hedge on uncertain claims, and follow instructions politely can also make it more cautious, more verbose, or slightly worse at raw problem-solving than an unaligned version of the same model. That gap — helpfulness or performance lost in exchange for safety and predictability — is the tax. For SaaS builders it shows up as over-refusals (the model declines a perfectly legitimate request because it pattern-matches to something risky), unnecessary disclaimers, or watered-down answers. The practical response isn't to strip safety out; it's to notice when alignment behavior is hurting your use case and address it with clear system prompts, careful prompt design, or a model tier tuned for your domain. Vendors work to shrink the alignment tax over time, so a model that over-refused last year may handle the same prompt fine today — re-test rather than assume the old behavior still holds.","The alignment tax is the capability you trade away to make a model safer — more caution, more hedging, more verbosity, slightly weaker raw problem-solving.",null,[11,14,17,20,23],{"slug":12,"name":13},"guardrails","Guardrails",{"slug":15,"name":16},"instruction-tuning","Instruction Tuning",{"slug":18,"name":19},"red-teaming","Red-Teaming",{"slug":21,"name":22},"rlhf","Reinforcement Learning from Human Feedback (RLHF)",{"slug":24,"name":25},"system-prompt","System Prompt",[27,31,35,38,41,45,48,51,54,57,60,63],{"slug":28,"category":5,"name":29,"updated_at":30},"agentic","Agentic AI","2026-08-24T02:46:36+00:00",{"slug":32,"category":5,"name":33,"updated_at":34},"artificial-intelligence","Artificial Intelligence (AI)","2026-08-24T02:46:38+00:00",{"slug":36,"category":5,"name":37,"updated_at":30},"attention","Attention",{"slug":39,"category":5,"name":40,"updated_at":34},"beam-search","Beam Search",{"slug":42,"category":5,"name":43,"updated_at":44},"benchmark-contamination","Benchmark Contamination","2026-08-24T02:46:37+00:00",{"slug":46,"category":5,"name":47,"updated_at":44},"catastrophic-forgetting","Catastrophic Forgetting",{"slug":49,"category":5,"name":50,"updated_at":34},"computer-vision","Computer Vision",{"slug":52,"category":5,"name":53,"updated_at":44},"constitutional-ai","Constitutional AI",{"slug":55,"category":5,"name":56,"updated_at":30},"context-window","Context Window",{"slug":58,"category":5,"name":59,"updated_at":34},"deep-learning","Deep Learning",{"slug":61,"category":5,"name":62,"updated_at":30},"diffusion-model","Diffusion Model",{"slug":64,"category":5,"name":65,"updated_at":44},"direct-preference-optimization","Direct Preference Optimization (DPO)"]