[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"glossary-ab-testing::en":3,"gloss-cluster-ab-testing::en":26,"gloss-next-ab-testing::en":9},{"slug":4,"category":5,"name":6,"definition":7,"meta_desc":8,"faq":9,"schema_markup":9,"related":10},"ab-testing","analytics","A\u002FB Testing","An A\u002FB test compares two versions of an experience by randomly assigning users to one of them and measuring a predefined metric. Randomisation is what makes the comparison causal: because assignment is independent of who the user is, any systematic difference in outcome is attributable to the change rather than to the two groups being different to begin with. That is the whole value, and it is also what before-and-after comparisons cannot give you, since seasonality, marketing activity and product changes all move numbers on their own. Running one honestly requires a few commitments made before launch. Pick a single primary metric and the minimum effect worth detecting, and use those to compute how long the test must run — a test whose duration is decided after the data comes in is a search for a favourable stopping point. Run in whole business cycles, usually complete weeks, since weekday and weekend traffic behave differently. Decide up front which secondary metrics act as guardrails, so a lift in conversion that comes with a rise in refunds is not recorded as a win. And check that assignment actually worked: unequal group sizes are the usual sign of an instrumentation bug that invalidates the result. The main practical limit is traffic. Most changes produce small effects, and small effects need large samples, so low-traffic products often cannot resolve the differences they care about — in which case sequencing changes, testing bigger swings, or relying on qualitative evidence is more honest than running an underpowered test and reading the result.","An A\u002FB test randomly splits users to attribute a metric change to the change itself — the commitments to fix before launch and why low traffic limits what you learn.",null,[11,14,17,20,23],{"slug":12,"name":13},"champion-challenger","Champion-Challenger (A\u002FB Model Testing)",{"slug":15,"name":16},"conversion-rate-optimization","Conversion Rate Optimization (CRO)",{"slug":18,"name":19},"feature-flag","Feature Flag",{"slug":21,"name":22},"product-analytics","Product Analytics",{"slug":24,"name":25},"statistical-significance","Statistical Significance",[27,31,35,38,41,44,47,50,53,56,59,62],{"slug":28,"category":5,"name":29,"updated_at":30},"active-user","Active User (DAU, WAU, MAU)","2026-08-24T02:46:38+00:00",{"slug":32,"category":5,"name":33,"updated_at":34},"autocapture","Autocapture","2026-08-24T02:46:37+00:00",{"slug":36,"category":5,"name":37,"updated_at":30},"cost-per-resolution","Cost per Resolution",{"slug":39,"category":5,"name":40,"updated_at":34},"customer-data-platform","Customer Data Platform (CDP)",{"slug":42,"category":5,"name":43,"updated_at":30},"deflection-rate","Deflection Rate",{"slug":45,"category":5,"name":46,"updated_at":30},"guardrail-metric","Guardrail Metric",{"slug":48,"category":5,"name":49,"updated_at":34},"identity-resolution","Identity Resolution",{"slug":51,"category":5,"name":52,"updated_at":34},"multi-touch-attribution","Multi-Touch Attribution",{"slug":54,"category":5,"name":55,"updated_at":30},"novelty-effect","Novelty Effect",{"slug":57,"category":5,"name":58,"updated_at":34},"retention-curve","Retention Curve",{"slug":60,"category":5,"name":61,"updated_at":30},"sample-ratio-mismatch","Sample Ratio Mismatch (SRM)",{"slug":63,"category":5,"name":64,"updated_at":30},"seat-utilization","Seat Utilization"]