[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"glossary-edge-ai::en":3,"gloss-cluster-edge-ai::en":23,"gloss-next-edge-ai::en":9},{"slug":4,"category":5,"name":6,"definition":7,"meta_desc":8,"faq":9,"schema_markup":9,"related":10},"edge-ai","cloud","Edge AI","Edge AI runs model inference on or near the device that produces the data — phones, browsers, IoT sensors, factory gateways — instead of round-tripping to a cloud API. The wins are latency (no network hop, so sub-50ms responses are feasible), privacy (raw data never leaves the device), offline resilience, and zero per-request inference bills. The constraint is capability: edge hardware fits quantized small language models, distilled vision models, and speech models, not frontier LLMs, so accuracy trails the cloud. Apple Intelligence's on-device tier, Gemini Nano on Android, and browser inference via WebGPU and ONNX Runtime are the mainstream examples. The dominant production pattern is hybrid: handle common, latency-sensitive, or sensitive requests locally and escalate hard ones to a cloud model. For SaaS builders, edge AI is also a pricing story — shifting inference cost from your margin onto hardware the customer already owns.","Edge AI runs inference on-device instead of in the cloud — lower latency, better privacy, offline support, and zero per-request model costs.",null,[11,14,17,20],{"slug":12,"name":13},"inference","Inference",{"slug":15,"name":16},"latency","Latency",{"slug":18,"name":19},"quantization","Quantization",{"slug":21,"name":22},"small-language-model","Small Language Model (SLM)",[24,28,31,34,38,41,44,47,50,53,56,59],{"slug":25,"category":5,"name":26,"updated_at":27},"autoscaling","Autoscaling","2026-08-24T02:46:37+00:00",{"slug":29,"category":5,"name":30,"updated_at":27},"availability-zone","Availability Zone (AZ)",{"slug":32,"category":5,"name":33,"updated_at":27},"block-storage","Block Storage",{"slug":35,"category":5,"name":36,"updated_at":37},"disaster-recovery","Disaster Recovery","2026-08-24T02:46:38+00:00",{"slug":39,"category":5,"name":40,"updated_at":27},"egress-fees","Egress Fees (Data Transfer Out)",{"slug":42,"category":5,"name":43,"updated_at":27},"finops","FinOps (Cloud Financial Operations)",{"slug":45,"category":5,"name":46,"updated_at":37},"immutable-infrastructure","Immutable Infrastructure",{"slug":48,"category":5,"name":49,"updated_at":37},"infrastructure-drift","Infrastructure Drift",{"slug":51,"category":5,"name":52,"updated_at":27},"managed-kubernetes","Managed Kubernetes",{"slug":54,"category":5,"name":55,"updated_at":27},"multi-region","Multi-Region",{"slug":57,"category":5,"name":58,"updated_at":37},"noisy-neighbor","Noisy Neighbor",{"slug":60,"category":5,"name":61,"updated_at":27},"platform-as-a-service","Platform as a Service (PaaS)"]