[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"glossary-pretraining::en":3,"gloss-cluster-pretraining::en":26,"gloss-next-pretraining::en":9},{"slug":4,"category":5,"name":6,"definition":7,"meta_desc":8,"faq":9,"schema_markup":9,"related":10},"pretraining","core-ai","Pretraining","Pretraining is the first and by far the most expensive stage of building a language model: the base network is trained on a very large, broad corpus with a simple objective — predict the next token — until it has absorbed grammar, facts, code, and a great deal of latent reasoning ability. Everything that comes later (instruction tuning, preference optimisation, safety alignment) reshapes behaviour that pretraining already installed; none of it adds knowledge at anything like the same scale. That ordering explains several things builders run into. A model cannot reliably answer questions about material that was never in its pretraining data, which is why a knowledge cutoff exists and why retrieval is the standard fix rather than more fine-tuning. Fine-tuning changes style, format and task-following far more easily than it changes what the model knows, because a fine-tuning set is orders of magnitude smaller than a pretraining corpus. And the cost profile is lopsided: pretraining is a capital project measured in cluster-months, while adapting a pretrained model is something a small team can do. For almost every SaaS product the practical decision is not whether to pretrain — it is which pretrained base to build on, and how much of the remaining gap to close with prompting, retrieval, or a light adaptation pass. Pretraining from scratch is justified mainly when the domain vocabulary is genuinely unlike public text, or when the data cannot leave the building at all.","Pretraining is the first, most expensive training stage where a base model learns from a broad corpus — and why fine-tuning changes style more than knowledge.",null,[11,14,17,20,23],{"slug":12,"name":13},"fine-tuning","Fine-Tuning",{"slug":15,"name":16},"foundation-model","Foundation Model",{"slug":18,"name":19},"instruction-tuning","Instruction Tuning",{"slug":21,"name":22},"scaling-laws","Scaling Laws",{"slug":24,"name":25},"self-supervised-learning","Self-Supervised Learning",[27,31,35,39,42,45,48,51,54,57,60,63],{"slug":28,"category":5,"name":29,"updated_at":30},"agentic","Agentic AI","2026-08-24T02:46:36+00:00",{"slug":32,"category":5,"name":33,"updated_at":34},"alignment-tax","Alignment Tax","2026-08-24T02:46:37+00:00",{"slug":36,"category":5,"name":37,"updated_at":38},"artificial-intelligence","Artificial Intelligence (AI)","2026-08-24T02:46:38+00:00",{"slug":40,"category":5,"name":41,"updated_at":30},"attention","Attention",{"slug":43,"category":5,"name":44,"updated_at":38},"beam-search","Beam Search",{"slug":46,"category":5,"name":47,"updated_at":34},"benchmark-contamination","Benchmark Contamination",{"slug":49,"category":5,"name":50,"updated_at":34},"catastrophic-forgetting","Catastrophic Forgetting",{"slug":52,"category":5,"name":53,"updated_at":38},"computer-vision","Computer Vision",{"slug":55,"category":5,"name":56,"updated_at":34},"constitutional-ai","Constitutional AI",{"slug":58,"category":5,"name":59,"updated_at":30},"context-window","Context Window",{"slug":61,"category":5,"name":62,"updated_at":38},"deep-learning","Deep Learning",{"slug":64,"category":5,"name":65,"updated_at":30},"diffusion-model","Diffusion Model"]