Module 8 · Sampling, Estimation, and Statistical Inference Module demo

A/B Test Analyzer

A new checkout button.

6:26 clipUses lessons 71–80Watch on YouTube

Transcript

48 sentences · select one to jump there

Code lab

Run it yourself

The demo source in one language. Edit it, run TypeScript and Python right here, and compare with the expected output.

demo-ab-test-analyzer.ts
Start from GitHub
/**
 * Fintech Math Bootcamp · Module 08 demo · A/B Test Analyzer
 * A checkout team tests a new one-click "Pay in 3" button (variant B) against the current "Pay now"
 * button (control A). B converts a little better, but each B order carries a provider fee. Is the lift
 * real, and is it worth paying for?
 * Lessons 071–080: estimands and estimators, sampling distributions, bias and error, the law of large
 * numbers, the central limit theorem, standard error, confidence intervals, hypotheses, p-values,
 * errors and power, effect size, practical significance and multiple comparisons.
 * Every visitor is simulated with a seeded generator; all numbers are synthetic.
 */

// A small seeded generator (mulberry32) so every run produces the same visitors.
export function mulberry32(seed: number): () => number {
  let a = seed >>> 0;
  return () => {
    a = (a + 0x6d2b79f5) >>> 0;
    let t = a;
    t = Math.imul(t ^ (t >>> 15), t | 1);
    t ^= t + Math.imul(t ^ (t >>> 7), t | 61);
    return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
  };
}

// The synthetic shopper population: a quarter are logged-in returning customers who convert at 6%,
// the rest are new visitors who convert at 2%. Overall conversion for control A is therefore 3%.
export const RETURNING_SHARE = 0.25, RETURNING_RATE = 0.06, NEW_RATE = 0.02;
export type Visitor = {returning: boolean; converted: boolean};
export function visitor(rand: () => number, lift = 1): Visitor {
  const returning = rand() < RETURNING_SHARE;
  const p = (returning ? RETURNING_RATE : NEW_RATE) * lift;
  return {returning, converted: rand() < p};
}
export const visitors = (rand: () => number, n: number, lift = 1) => Array.from({length: n}, () => visitor(rand, lift));

const mean = (a: number[]) => a.reduce((s, v) => s + v, 0) / a.length;
const rate = (vs: Visitor[]) => vs.filter(v => v.converted).length / vs.length;

// 071 · the estimand is a population question; the estimator is a recipe; the estimate is one number
export const differenceInProportions = (convA: number, nA: number, convB: number, nB: number) => convB / nB - convA / nA;

// 074 · law of large numbers: the running conversion rate after each visitor
export function runningRate(vs: Visitor[], checkpoints: number[]): number[] {
  let conv = 0, k = 0;
  const out: number[] = [];
  vs.forEach((v, i) => {
    if (v.converted) conv++;
    while (k < checkpoints.length && checkpoints[k] === i + 1) {out.push(conv / (i + 1)); k++;}
  });
  return out;
}

// 072 · sampling distribution: the same estimator applied to many independent samples
export const sampleRates = (rand: () => number, samples: number, n: number) => Array.from({length: samples}, () => rate(visitors(rand, n)));

// 073 · bias, variance and mean squared error of an estimator against a known truth (MSE = variance + bias²)
export function estimatorError(estimates: number[], truth: number) {
  const m = mean(estimates), bias = m - truth;
  const variance = mean(estimates.map(x => (x - m) ** 2));
  const mse = mean(estimates.map(x => (x - truth) ** 2));
  return {mean: m, bias, variance, mse, rmse: Math.sqrt(mse)};
}

// 075 · central limit theorem: standardize sample means, Z = √n(mean − μ)/σ
export const standardize = (sampleMean: number, mu: number, sigma: number, n: number) => Math.sqrt(n) * (sampleMean - mu) / sigma;

// histogram helper for drawing a sampling distribution
export function histogram(values: number[], lo: number, hi: number, bins: number): number[] {
  const h = new Array<number>(bins).fill(0);
  for (const v of values) {const b = Math.floor((v - lo) / (hi - lo) * bins + 1e-9); if (b >= 0 && b < bins) h[b]++;}
  return h;
}

// 076 · standard error of a difference in two independent proportions
export const seDifference = (pA: number, nA: number, pB: number, nB: number) => Math.sqrt(pA * (1 - pA) / nA + pB * (1 - pB) / nB);

// Standard normal CDF via the Abramowitz and Stegun 7.1.26 approximation of erf (absolute error below 1.5e-7).
export function normalCdf(z: number): number {
  const x = Math.abs(z) / Math.SQRT2, t = 1 / (1 + 0.3275911 * x);
  const erf = 1 - ((((1.061405429 * t - 1.453152027) * t + 1.421413741) * t - 0.284496736) * t + 0.254829592) * t * Math.exp(-x * x);
  return z >= 0 ? 0.5 * (1 + erf) : 0.5 * (1 - erf);
}
// Normal quantile by bisection on normalCdf (accurate to about 1e-9 in z, inheriting the CDF approximation).
export function normalQuantile(p: number): number {
  let lo = -10, hi = 10;
  for (let i = 0; i < 80; i++) {const mid = (lo + hi) / 2; if (normalCdf(mid) < p) lo = mid; else hi = mid;}
  return (lo + hi) / 2;
}

// 077 · confidence interval: estimate ± critical × SE
export const confidenceInterval = (estimate: number, se: number, level = 0.95) => {
  const critical = normalQuantile(1 - (1 - level) / 2);
  return {low: estimate - critical * se, high: estimate + critical * se, critical};
};

// 078 · H0: no difference in conversion; test statistic uses the pooled rate under H0
export function zTest(convA: number, nA: number, convB: number, nB: number) {
  const pooled = (convA + convB) / (nA + nB);
  const se0 = Math.sqrt(pooled * (1 - pooled) * (1 / nA + 1 / nB));
  const z = differenceInProportions(convA, nA, convB, nB) / se0;
  return {pooled, se0, z, pValue: 2 * (1 - normalCdf(Math.abs(z)))}; // 079 · two-sided p-value
}

// 079 · power: probability of rejecting H0 when the true difference is delta (normal approximation)
export function power(pA: number, delta: number, n: number, alpha = 0.05) {
  const zc = normalQuantile(1 - alpha / 2), se = seDifference(pA, n, pA + delta, n);
  return normalCdf(delta / se - zc) + normalCdf(-delta / se - zc);
}
export function sampleSizeFor(pA: number, delta: number, targetPower = 0.8, alpha = 0.05) {
  const za = normalQuantile(1 - alpha / 2), zb = normalQuantile(targetPower);
  return Math.ceil((za + zb) ** 2 * (pA * (1 - pA) + (pA + delta) * (1 - pA - delta)) / delta ** 2);
}

// 080 · effect size standardized by the per-visitor standard deviation, and Bonferroni's α/m
export const standardizedEffect = (diff: number, pA: number) => diff / Math.sqrt(pA * (1 - pA));
export const bonferroni = (alpha: number, tests: number) => alpha / tests;
export const familywiseError = (alpha: number, tests: number) => 1 - (1 - alpha) ** tests;

export function runDemo() {
  const rand = mulberry32(8081);
  const n = 60000, trueA = RETURNING_SHARE * RETURNING_RATE + (1 - RETURNING_SHARE) * NEW_RATE, trueLift = 1.1;
  const trueB = trueA * trueLift;

  // the experiment: 60,000 visitors per arm
  const armA = visitors(rand, n), armB = visitors(rand, n, trueLift);
  const convA = armA.filter(v => v.converted).length, convB = armB.filter(v => v.converted).length;
  const pA = convA / n, pB = convB / n, diff = differenceInProportions(convA, n, convB, n);
  const checkpoints = [20, 50, 100, 200, 300, 500, 750, 1000, 1500, 2000, 3000, 4000, 5000, 7500, 10000, 15000, 20000, 30000, 40000, 50000, 60000];
  const loggedInOnlyA = rate(armA.filter(v => v.returning));

  // 400 repeated samples of the control population, at two sample sizes
  const small = sampleRates(rand, 400, 100);
  const bigSamples = Array.from({length: 400}, () => visitors(rand, 2000));
  const big = bigSamples.map(rate);
  const sigma = Math.sqrt(trueA * (1 - trueA));
  const zBig = big.map(m => standardize(m, trueA, sigma, 2000));
  const estimators = {
    fullSample: estimatorError(big, trueA),
    first200: estimatorError(bigSamples.map(s => rate(s.slice(0, 200))), trueA),
    loggedInOnly: estimatorError(bigSamples.map(s => rate(s.filter(v => v.returning))), trueA),
  };

  // inference on the real experiment
  const se = seDifference(pA, n, pB, n), ci = confidenceInterval(diff, se), test = zTest(convA, n, convB, n);

  // A/A tests: both arms identical, so every rejection is a false positive
  const aa = bigSamples.map(s => {const other = visitors(rand, 2000), cA = s.filter(v => v.converted).length, cB = other.filter(v => v.converted).length; return cA + cB === 0 ? 1 : zTest(cA, 2000, cB, 2000).pValue;});
  const aaRejected = aa.filter(p => p < 0.05).length, aaRejectedIdx = aa.flatMap((p, i) => (p < 0.05 ? [i] : []));
  const launches = Array.from({length: 20}, (_, k) => aa.slice(k * 20, k * 20 + 20));
  const launchesWithFalseWinner = launches.filter(l => l.some(p => p < 0.05)).length;

  // economics: 12 dollars of margin per order; each B order pays a 2 dollar provider fee
  const margin = 12, fee = 2;
  const profitA = pA * margin, profitB = pB * (margin - fee);
  const breakEvenRate = pA * margin / (margin - fee), breakEvenLift = breakEvenRate - pA;
  const mde = 0.0025;
  const nGrid = [10000, 20000, 40000, 60000, 80000, 120000, 160000, 200000];

  return {
    setup: {visitorsPerArm: n, trueA, trueB, returningShare: RETURNING_SHARE, returningRate: RETURNING_RATE, newRate: NEW_RATE},
    experiment: {convA, convB, pA, pB, diff, relativeLift: diff / pA},
    lln: {checkpoints, runningA: runningRate(armA, checkpoints), runningB: runningRate(armB, checkpoints), loggedInOnlyA},
    sampling: {
      smallN: 100, bigN: 2000, samples: 400,
      smallHist: histogram(small, -0.005, 0.095, 10), bigHist: histogram(big, 0.02, 0.04, 20),
      smallMean: mean(small), bigMean: mean(big), bigSd: Math.sqrt(estimatorError(big, trueA).variance),
      theorySe: sigma / Math.sqrt(2000), zWithin196: zBig.filter(z => Math.abs(z) <= 1.96).length / zBig.length,
      estimators,
    },
    inference: {se, ci, ...test, alpha: 0.05, reject: test.pValue < 0.05},
    errors: {aaTests: aa.length, aaRejected, aaRejectedIdx, aaRate: aaRejected / aa.length, mde, powerAtMde: power(pA, mde, n), nFor80: sampleSizeFor(pA, mde),
      powerCurve: nGrid.map(k => ({n: k, power: power(pA, mde, k)}))},
    practical: {
      margin, fee, profitA, profitB, per100k: {A: profitA * 100000, B: profitB * 100000},
      breakEvenRate, breakEvenLift, ciBelowBreakEven: ci.high < breakEvenLift,
      standardizedEffect: standardizedEffect(diff, pA),
    },
    multiple: {variants: 20, familywise: familywiseError(0.05, 20), bonferroni: bonferroni(0.05, 20), launches: launches.length, launchesWithFalseWinner,
      passesBonferroni: test.pValue < bonferroni(0.05, 20)},
  };
}

export const checkedResult = {"setup":{"visitorsPerArm":60000,"trueA":0.03,"trueB":0.033,"returningShare":0.25,"returningRate":0.06,"newRate":0.02},"experiment":{"convA":1819,"convB":1977,"pA":0.030316666666666665,"pB":0.03295,"diff":0.0026333333333333347,"relativeLift":0.08686091258933484},"lln":{"checkpoints":[20,50,100,200,300,500,750,1000,1500,2000,3000,4000,5000,7500,10000,15000,20000,30000,40000,50000,60000],"runningA":[0,0.08,0.06,0.06,0.04666666666666667,0.04,0.034666666666666665,0.029,0.027333333333333334,0.028,0.028,0.028,0.0288,0.029866666666666666,0.0298,0.0296,0.03,0.0306,0.03055,0.03004,0.030316666666666665],"runningB":[0.05,0.04,0.02,0.02,0.03666666666666667,0.044,0.036,0.036,0.037333333333333336,0.0345,0.035,0.03425,0.0324,0.0324,0.0323,0.03166666666666667,0.0317,0.03213333333333333,0.032425,0.03292,0.03295],"loggedInOnlyA":0.059108331652208995},"sampling":{"smallN":100,"bigN":2000,"samples":400,"smallHist":[17,73,87,99,70,23,22,4,4,0],"bigHist":[1,4,2,6,12,18,27,42,49,37,37,38,38,24,19,17,11,5,6,4],"smallMean":0.02852499999999983,"bigMean":0.03007000000000002,"bigSd":0.0037659792883126667,"theorySe":0.00381444622455213,"zWithin196":0.945,"estimators":{"fullSample":{"mean":0.03007000000000002,"bias":0.00007000000000002143,"variance":0.00001418259999999998,"mse":0.00001418750000000002,"rmse":0.0037666297933298433},"first200":{"mean":0.03032499999999995,"bias":0.00032499999999995033,"variance":0.00013526937499999995,"mse":0.00013537500000000008,"rmse":0.0116350762782201},"loggedInOnly":{"mean":0.060012718133932205,"bias":0.030012718133932206,"variance":0.00011267280822335788,"mse":0.0010134360580102216,"rmse":0.031834510487994336}}},"inference":{"se":0.001010460818050267,"ci":{"low":0.0006528677157883146,"high":0.004613798950878355,"critical":1.9599628032746725},"pooled":0.03163333333333333,"se0":0.0010104894120434177,"z":2.60599794708209,"pValue":0.009160771053956296,"alpha":0.05,"reject":true},"errors":{"aaTests":400,"aaRejected":20,"aaRejectedIdx":[81,82,115,122,140,174,191,194,202,221,226,260,271,278,287,288,328,351,363,391],"aaRate":0.05,"mde":0.0025,"powerAtMde":0.6973130794984782,"nFor80":76778,"powerCurve":[{"n":10000,"power":0.1728249135307212},{"n":20000,"power":0.29837956265396465},{"n":40000,"power":0.5248338656151287},{"n":60000,"power":0.6973130794984782},{"n":80000,"power":0.8158906675621123},{"n":120000,"power":0.9385278287807004},{"n":160000,"power":0.9814367979007479},{"n":200000,"power":0.9947924895252089}]},"practical":{"margin":12,"fee":2,"profitA":0.3638,"profitB":0.3295,"per100k":{"A":36380,"B":32950},"breakEvenRate":0.03638,"breakEvenLift":0.006063333333333337,"ciBelowBreakEven":true,"standardizedEffect":0.015358547551216697},"multiple":{"variants":20,"familywise":0.6415140775914581,"bonferroni":0.0025,"launches":20,"launchesWithFalseWinner":14,"passesBonferroni":false}};

// Run this file directly: npx tsx lessons/08-sampling-estimation-and-statistical-inference/demo-ab-test-analyzer.ts
if (process.argv[1] && import.meta.url.endsWith(process.argv[1].replace(/\\/g, "/").split("/").pop()!)) {
  console.log(JSON.stringify(runDemo(), null, 2));
}

Your output

Press Run to execute the code in your browser.

Expected output

{
  "setup": {
    "visitorsPerArm": 60000,
    "trueA": 0.03,
    "trueB": 0.033,
    "returningShare": 0.25,
    "returningRate": 0.06,
    "newRate": 0.02
  },
  "experiment": {
    "convA": 1819,
    "convB": 1977,
    "pA": 0.030316666666666665,
    "pB": 0.03295,
    "diff": 0.0026333333333333347,
    "relativeLift": 0.08686091258933484
  },
  "lln": {
    "checkpoints": [
      20,
      50,
      100,
      200,
      300,
      500,
      750,
      1000,
      1500,
      2000,
      3000,
      4000,
      5000,
      7500,
      10000,
      15000,
      20000,
      30000,
      40000,
      50000,
      60000
    ],
    "runningA": [
      0,
      0.08,
      0.06,
      0.06,
      0.04666666666666667,
      0.04,
      0.034666666666666665,
      0.029,
      0.027333333333333334,
      0.028,
      0.028,
      0.028,
      0.0288,
      0.029866666666666666,
      0.0298,
      0.0296,
      0.03,
      0.0306,
      0.03055,
      0.03004,
      0.030316666666666665
    ],
    "runningB": [
      0.05,
      0.04,
      0.02,
      0.02,
      0.03666666666666667,
      0.044,
      0.036,
      0.036,
      0.037333333333333336,
      0.0345,
      0.035,
      0.03425,
      0.0324,
      0.0324,
      0.0323,
      0.03166666666666667,
      0.0317,
      0.03213333333333333,
      0.032425,
      0.03292,
      0.03295
    ],
    "loggedInOnlyA": 0.059108331652208995
  },
  "sampling": {
    "smallN": 100,
    "bigN": 2000,
    "samples": 400,
    "smallHist": [
      17,
      73,
      87,
      99,
      70,
      23,
      22,
      4,
      4,
      0
    ],
    "bigHist": [
      1,
      4,
      2,
      6,
      12,
      18,
      27,
      42,
      49,
      37,
      37,
      38,
      38,
      24,
      19,
      17,
      11,
      5,
      6,
      4
    ],
    "smallMean": 0.02852499999999983,
    "bigMean": 0.03007000000000002,
    "bigSd": 0.0037659792883126667,
    "theorySe": 0.00381444622455213,
    "zWithin196": 0.945,
    "estimators": {
      "fullSample": {
        "mean": 0.03007000000000002,
        "bias": 0.00007000000000002143,
        "variance": 0.00001418259999999998,
        "mse": 0.00001418750000000002,
        "rmse": 0.0037666297933298433
      },
      "first200": {
        "mean": 0.03032499999999995,
        "bias": 0.00032499999999995033,
        "variance": 0.00013526937499999995,
        "mse": 0.00013537500000000008,
        "rmse": 0.0116350762782201
      },
      "loggedInOnly": {
        "mean": 0.060012718133932205,
        "bias": 0.030012718133932206,
        "variance": 0.00011267280822335788,
        "mse": 0.0010134360580102216,
        "rmse": 0.031834510487994336
      }
    }
  },
  "inference": {
    "se": 0.001010460818050267,
    "ci": {
      "low": 0.0006528677157883146,
      "high": 0.004613798950878355,
      "critical": 1.9599628032746725
    },
    "pooled": 0.03163333333333333,
    "se0": 0.0010104894120434177,
    "z": 2.60599794708209,
    "pValue": 0.009160771053956296,
    "alpha": 0.05,
    "reject": true
  },
  "errors": {
    "aaTests": 400,
    "aaRejected": 20,
    "aaRejectedIdx": [
      81,
      82,
      115,
      122,
      140,
      174,
      191,
      194,
      202,
      221,
      226,
      260,
      271,
      278,
      287,
      288,
      328,
      351,
      363,
      391
    ],
    "aaRate": 0.05,
    "mde": 0.0025,
    "powerAtMde": 0.6973130794984782,
    "nFor80": 76778,
    "powerCurve": [
      {
        "n": 10000,
        "power": 0.1728249135307212
      },
      {
        "n": 20000,
        "power": 0.29837956265396465
      },
      {
        "n": 40000,
        "power": 0.5248338656151287
      },
      {
        "n": 60000,
        "power": 0.6973130794984782
      },
      {
        "n": 80000,
        "power": 0.8158906675621123
      },
      {
        "n": 120000,
        "power": 0.9385278287807004
      },
      {
        "n": 160000,
        "power": 0.9814367979007479
      },
      {
        "n": 200000,
        "power": 0.9947924895252089
      }
    ]
  },
  "practical": {
    "margin": 12,
    "fee": 2,
    "profitA": 0.3638,
    "profitB": 0.3295,
    "per100k": {
      "A": 36380,
      "B": 32950
    },
    "breakEvenRate": 0.03638,
    "breakEvenLift": 0.006063333333333337,
    "ciBelowBreakEven": true,
    "standardizedEffect": 0.015358547551216697
  },
  "multiple": {
    "variants": 20,
    "familywise": 0.6415140775914581,
    "bonferroni": 0.0025,
    "launches": 20,
    "launchesWithFalseWinner": 14,
    "passesBonferroni": false
  }
}

Prefer your own machine? Every file is in the course repository · open it in Codespaces.

What the demo does

A checkout team tests a one-click "Pay in 3" button against "Pay now" on 60,000 simulated visitors per group. The lift is statistically significant, but too small to pay the new button's fee. Every module 08 lesson becomes one analyzer feature.