Module 9 · Dependence, Regression, and Model Foundations Module demo

Credit Limit Scoring Model

Income, spending and a limit.

6:35 clipUses lessons 81–90Watch on YouTube

Transcript

41 sentences · select one to jump there

Code lab

Run it yourself

The demo source in one language. Edit it, run TypeScript and Python right here, and compare with the expected output.

demo-credit-limit-scoring-model.ts
Start from GitHub
/**
 * Fintech Math Bootcamp · Module 09 demo · Credit Limit Scoring Model
 * A card issuer's prototype suggests a starting credit limit from monthly income: fit monthly card
 * spending on income, then apply a policy of three months of predicted spending. How strong is the
 * relationship, how good is the fit, and where does the model stop being trustworthy?
 * Lessons 081–090: scatter plots, covariance, Pearson, Spearman and Kendall, confounding, simple OLS,
 * interpreting coefficients, residual metrics, R² and adjusted R², heteroskedasticity and
 * multicollinearity.
 * Every applicant is synthetic, generated from a seeded formula. This is a teaching model, not a
 * description of how any lender sets credit limits.
 */

export type Applicant = {income: number; spend: number};

// Seeded generator (mulberry32) and Box–Muller normal draws, so the data are identical on every run.
export function mulberry32(seed: number): () => number {
  let a = seed >>> 0;
  return () => {
    a = (a + 0x6d2b79f5) >>> 0;
    let t = a;
    t = Math.imul(t ^ (t >>> 15), t | 1);
    t ^= t + Math.imul(t ^ (t >>> 7), t | 61);
    return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
  };
}
const normal = (rand: () => number) => Math.sqrt(-2 * Math.log(1 - rand())) * Math.cos(2 * Math.PI * rand());

// The hidden truth of this synthetic world: spending grows with income but flattens out (concave),
// and its noise widens as income rises (heteroskedastic).
export const trueSpend = (income: number) => 600 + 900 * Math.log(income / 1000);
export const noiseSd = (income: number) => 0.06 * income - 40;

const mean = (a: number[]) => a.reduce((s, v) => s + v, 0) / a.length;
const dot = (a: number[], b: number[]) => a.reduce((s, v, i) => s + v * b[i], 0);
const centered = (a: number[]) => {const m = mean(a); return a.map(v => v - m);};

// 082 · sample covariance: Σ(x − x̄)(y − ȳ) / (n − 1)
export const covariance = (x: number[], y: number[]) => dot(centered(x), centered(y)) / (x.length - 1);

// 083 · Pearson correlation: Sxy / √(Sxx Syy)
export function pearson(x: number[], y: number[]): number {
  const dx = centered(x), dy = centered(y), sxx = dot(dx, dx), syy = dot(dy, dy);
  if (sxx === 0 || syy === 0) throw new Error("Correlation undefined: zero variance");
  return dot(dx, dy) / Math.sqrt(sxx * syy);
}

// 084 · Spearman = Pearson on ranks (ties get average ranks); Kendall tau-a = (concordant − discordant) / pairs
export const ranks = (a: number[]) => a.map(v => 1 + a.filter(z => z < v).length + (a.filter(z => z === v).length - 1) / 2);
export const spearman = (x: number[], y: number[]) => pearson(ranks(x), ranks(y));
export function kendallTauA(x: number[], y: number[]): number {
  let s = 0;
  for (let i = 0; i < x.length; i++) for (let j = i + 1; j < x.length; j++) s += Math.sign((x[j] - x[i]) * (y[j] - y[i]));
  return s / (x.length * (x.length - 1) / 2);
}

// 085 · a pooled comparison that reverses inside every group (Simpson's paradox)
export type Cell = {onTime: number; total: number};
export function simpson(a: {low: Cell; high: Cell}, b: {low: Cell; high: Cell}) {
  const r = (c: Cell) => c.onTime / c.total;
  const pool = (g: {low: Cell; high: Cell}) => (g.low.onTime + g.high.onTime) / (g.low.total + g.high.total);
  return {pooledA: pool(a), pooledB: pool(b), lowRisk: [r(a.low), r(b.low)], highRisk: [r(a.high), r(b.high)]};
}

// 086 · simple OLS: b1 = Sxy / Sxx, b0 = ȳ − b1 x̄
export function ols(x: number[], y: number[]) {
  const dx = centered(x), sxx = dot(dx, dx);
  if (sxx === 0) throw new Error("Slope undefined: no spread in x");
  const slope = dot(dx, centered(y)) / sxx, intercept = mean(y) - slope * mean(x);
  return {slope, intercept, xMin: Math.min(...x), xMax: Math.max(...x)};
}

// 087 · prediction = intercept + slope · input, flagged when the input is outside the fitted range
export const predict = (m: {slope: number; intercept: number}, x: number) => m.intercept + m.slope * x;
export const outsideFit = (m: {xMin: number; xMax: number}, x: number) => x < m.xMin || x > m.xMax;
// policy used by this prototype: a starting limit of three months of predicted spending, rounded to 100
export const suggestedLimit = (predictedSpend: number) => Math.round(3 * predictedSpend / 100) * 100;

// 088 · residuals and error metrics
export function errorMetrics(actual: number[], predicted: number[]) {
  const e = actual.map((v, i) => v - predicted[i]);
  const mse = mean(e.map(v => v * v));
  return {residuals: e, mae: mean(e.map(Math.abs)), mse, rmse: Math.sqrt(mse)};
}

// 089 · R² against the mean-only baseline, and adjusted R² for p predictors
export function rSquared(actual: number[], predicted: number[], p: number) {
  const sse = dot(actual.map((v, i) => v - predicted[i]), actual.map((v, i) => v - predicted[i]));
  const dy = centered(actual), sst = dot(dy, dy), n = actual.length;
  if (sst === 0) throw new Error("R-squared undefined: constant target");
  if (n - p - 1 <= 0) throw new Error("Adjusted R-squared needs n > p+1");
  const r2 = 1 - sse / sst;
  return {sse, sst, r2, adjusted: 1 - (1 - r2) * (n - 1) / (n - p - 1)};
}

// two-predictor OLS through the centered normal equations (used for the extra-predictor comparison)
export function ols2(x1: number[], x2: number[], y: number[]) {
  const a = centered(x1), b = centered(x2), c = centered(y);
  const s11 = dot(a, a), s22 = dot(b, b), s12 = dot(a, b), s1y = dot(a, c), s2y = dot(b, c);
  const det = s11 * s22 - s12 * s12;
  const scale = s11 * s22; // determinant relative to its largest possible size
  if (Math.abs(det) <= 1e-12 * scale) return {identified: false as const, relativeDet: det / scale};
  const b1 = (s22 * s1y - s12 * s2y) / det, b2 = (s11 * s2y - s12 * s1y) / det;
  const b0 = mean(y) - b1 * mean(x1) - b2 * mean(x2);
  return {identified: true as const, b0, b1, b2, relativeDet: det / scale, predictions: x1.map((v, i) => b0 + b1 * v + b2 * x2[i])};
}

// 090 · heteroskedasticity: mean squared residual in the lower and upper halves of income
export function spreadByHalf(x: number[], residuals: number[]) {
  const order = x.map((v, i) => [v, i]).sort((p, q) => p[0] - q[0]).map(p => p[1]);
  const half = Math.floor(order.length / 2), ms = (idx: number[]) => mean(idx.map(i => residuals[i] ** 2));
  return {lowRms: Math.sqrt(ms(order.slice(0, half))), highRms: Math.sqrt(ms(order.slice(half)))};
}

export function runDemo() {
  const rand = mulberry32(916);
  const n = 30;
  // incomes spread evenly across 2,000–9,000 a month with a little jitter, rounded to 50
  const applicants: Applicant[] = Array.from({length: n}, (_, i) => {
    const income = Math.round((2000 + 7000 * (i + 0.5) / n + 150 * (rand() - 0.5)) / 50) * 50;
    return {income, spend: Math.round(trueSpend(income) + noiseSd(income) * normal(rand))};
  });
  const x = applicants.map(a => a.income), y = applicants.map(a => a.spend);

  // association
  const cov = covariance(x, y), r = pearson(x, y), rho = spearman(x, y), tau = kendallTauA(x, y);
  const dx = centered(x), dy = centered(y);
  const quadrants = {agree: dx.filter((v, i) => v * dy[i] > 0).length, disagree: dx.filter((v, i) => v * dy[i] < 0).length};

  // the model
  const model = ols(x, y), fitted = x.map(v => predict(model, v));
  const metrics = errorMetrics(y, fitted), fit = rSquared(y, fitted, 1);
  const example = {income: 5500, spend: predict(model, 5500), limit: suggestedLimit(predict(model, 5500))};

  // an extra predictor with no real information: the day of the month the application arrived
  const day = applicants.map(() => 1 + Math.floor(rand() * 28));
  const withDay = ols2(x, day, y);
  const fitWithDay = withDay.identified ? rSquared(y, withDay.predictions, 2) : null;
  // a redundant predictor: annual income is exactly 12 × monthly income
  const annual = x.map(v => 12 * v);
  const redundant = ols2(x, annual, y);
  const r2Between = pearson(x, annual) ** 2;

  // heteroskedasticity
  const spread = spreadByHalf(x, metrics.residuals);

  // extrapolation: three high earners far outside the fitted range, drawn from the same hidden truth
  const highEarners = [15000, 22000, 30000].map(income => {
    const spend = Math.round(trueSpend(income) + noiseSd(income) * normal(rand));
    const pred = predict(model, income);
    return {income, spend, predicted: pred, limit: suggestedLimit(pred), limitFromActual: suggestedLimit(spend), outside: outsideFit(model, income)};
  });
  // the in-sample error band the model would quote: ±1.96 × RMSE (it ignores extrapolation entirely)
  const band = 1.96 * metrics.rmse;
  const missedByBand = highEarners.filter(h => Math.abs(h.spend - h.predicted) > band).length;

  // confounding: on-time repayment under two limit policies, split by risk band
  const policyA = {low: {onTime: 450, total: 500}, high: {onTime: 30, total: 100}};
  const policyB = {low: {onTime: 95, total: 100}, high: {onTime: 175, total: 500}};

  return {
    applicants, n,
    association: {covariance: cov, pearson: r, spearman: rho, kendall: tau, quadrants, meanIncome: mean(x), meanSpend: mean(y)},
    confounding: {policyA, policyB, ...simpson(policyA, policyB)},
    model: {...model, slopePer1000: model.slope * 1000, example, interceptOutside: outsideFit(model, 0)},
    errors: {mae: metrics.mae, mse: metrics.mse, rmse: metrics.rmse, residuals: metrics.residuals,
      largest: metrics.residuals.reduce((m, e) => (Math.abs(e) > Math.abs(m) ? e : m), 0)},
    fit: {r2: fit.r2, adjusted: fit.adjusted, sse: fit.sse, sst: fit.sst,
      withDay: fitWithDay ? {r2: fitWithDay.r2, adjusted: fitWithDay.adjusted} : null},
    diagnostics: {lowRms: spread.lowRms, highRms: spread.highRms, ratio: spread.highRms / spread.lowRms,
      annualR2Between: r2Between, annualIdentified: redundant.identified, annualRelativeDet: redundant.relativeDet},
    extrapolation: {highEarners, fittedRange: [model.xMin, model.xMax], band, missedByBand, overshoot: highEarners.map(h => h.predicted - h.spend)},
  };
}

export const checkedResult = {"applicants":[{"income":2200,"spend":1252},{"income":2400,"spend":1443},{"income":2600,"spend":1410},{"income":2850,"spend":1487},{"income":3000,"spend":1794},{"income":3300,"spend":1676},{"income":3500,"spend":1847},{"income":3800,"spend":1639},{"income":3950,"spend":1551},{"income":4300,"spend":1838},{"income":4450,"spend":1910},{"income":4750,"spend":1790},{"income":4900,"spend":2334},{"income":5100,"spend":1742},{"income":5450,"spend":2074},{"income":5550,"spend":1797},{"income":5850,"spend":2166},{"income":6100,"spend":2712},{"income":6300,"spend":2796},{"income":6600,"spend":2293},{"income":6800,"spend":2583},{"income":7000,"spend":2565},{"income":7250,"spend":2982},{"income":7450,"spend":2852},{"income":7750,"spend":2932},{"income":7950,"spend":2073},{"income":8150,"spend":2705},{"income":8400,"spend":2966},{"income":8650,"spend":2072},{"income":8900,"spend":2470}],"n":30,"association":{"covariance":883210.0574712645,"pearson":0.8255346867188061,"spearman":0.8269187986651836,"kendall":0.6459770114942529,"quadrants":{"agree":26,"disagree":4},"meanIncome":5508.333333333333,"meanSpend":2125.0333333333333},"confounding":{"policyA":{"low":{"onTime":450,"total":500},"high":{"onTime":30,"total":100}},"policyB":{"low":{"onTime":95,"total":100},"high":{"onTime":175,"total":500}},"pooledA":0.8,"pooledB":0.45,"lowRisk":[0.9,0.95],"highRisk":[0.3,0.35]},"model":{"slope":0.2115728028360495,"intercept":959.6198110447606,"xMin":2200,"xMax":8900,"slopePer1000":211.5728028360495,"example":{"income":5500,"spend":2123.270226643033,"limit":6400},"interceptOutside":true},"errors":{"mae":224.8771150768383,"mse":84416.8457817779,"rmse":290.5457722662264,"residuals":[-173.07997728406963,-24.394537851279438,-99.70909841848925,-75.60229912750174,199.6617804470909,18.189939596275963,146.87537902906615,-124.59646182174856,-244.33238224715615,-31.382863239773542,8.8812163348191,-174.59062451599584,337.6734550585968,-296.641105508613,-38.69158650123063,-336.8488667848351,-31.32070763565025,461.78609165533726,503.47153108812745,-63.000309762687266,184.68512967010247,124.37056910289266,488.47736839388017,316.16280782667036,332.69096697585564,-568.6235935913542,21.06184584143557,229.16864513242353,-717.724555576589,-372.617756285601],"largest":-717.724555576589},"fit":{"r2":0.6815075189759174,"adjusted":0.6701327875107717,"sse":2532505.373453337,"sst":7951538.966666667,"withDay":{"r2":0.6885774625921738,"adjusted":0.6655091264878904}},"diagnostics":{"lowRms":166.98543760428234,"highRms":375.43249085776006,"ratio":2.2482947989000697,"annualR2Between":1.0000000000000004,"annualIdentified":false,"annualRelativeDet":-4.85213790726814e-16},"extrapolation":{"highEarners":[{"income":15000,"spend":2188,"predicted":4133.211853585503,"limit":12400,"limitFromActual":6600,"outside":true},{"income":22000,"spend":4352,"predicted":5614.221473437849,"limit":16800,"limitFromActual":13100,"outside":true},{"income":30000,"spend":5511,"predicted":7306.8038961262455,"limit":21900,"limitFromActual":16500,"outside":true}],"fittedRange":[2200,8900],"band":569.4697136418037,"missedByBand":3,"overshoot":[1945.2118535855034,1262.2214734378495,1795.8038961262455]}};

// Run this file directly: npx tsx lessons/09-dependence-regression-and-model-foundations/demo-credit-limit-scoring-model.ts
if (process.argv[1] && import.meta.url.endsWith(process.argv[1].replace(/\\/g, "/").split("/").pop()!)) {
  console.log(JSON.stringify(runDemo(), null, 2));
}

Your output

Press Run to execute the code in your browser.

Expected output

{
  "applicants": [
    {
      "income": 2200,
      "spend": 1252
    },
    {
      "income": 2400,
      "spend": 1443
    },
    {
      "income": 2600,
      "spend": 1410
    },
    {
      "income": 2850,
      "spend": 1487
    },
    {
      "income": 3000,
      "spend": 1794
    },
    {
      "income": 3300,
      "spend": 1676
    },
    {
      "income": 3500,
      "spend": 1847
    },
    {
      "income": 3800,
      "spend": 1639
    },
    {
      "income": 3950,
      "spend": 1551
    },
    {
      "income": 4300,
      "spend": 1838
    },
    {
      "income": 4450,
      "spend": 1910
    },
    {
      "income": 4750,
      "spend": 1790
    },
    {
      "income": 4900,
      "spend": 2334
    },
    {
      "income": 5100,
      "spend": 1742
    },
    {
      "income": 5450,
      "spend": 2074
    },
    {
      "income": 5550,
      "spend": 1797
    },
    {
      "income": 5850,
      "spend": 2166
    },
    {
      "income": 6100,
      "spend": 2712
    },
    {
      "income": 6300,
      "spend": 2796
    },
    {
      "income": 6600,
      "spend": 2293
    },
    {
      "income": 6800,
      "spend": 2583
    },
    {
      "income": 7000,
      "spend": 2565
    },
    {
      "income": 7250,
      "spend": 2982
    },
    {
      "income": 7450,
      "spend": 2852
    },
    {
      "income": 7750,
      "spend": 2932
    },
    {
      "income": 7950,
      "spend": 2073
    },
    {
      "income": 8150,
      "spend": 2705
    },
    {
      "income": 8400,
      "spend": 2966
    },
    {
      "income": 8650,
      "spend": 2072
    },
    {
      "income": 8900,
      "spend": 2470
    }
  ],
  "n": 30,
  "association": {
    "covariance": 883210.0574712645,
    "pearson": 0.8255346867188061,
    "spearman": 0.8269187986651836,
    "kendall": 0.6459770114942529,
    "quadrants": {
      "agree": 26,
      "disagree": 4
    },
    "meanIncome": 5508.333333333333,
    "meanSpend": 2125.0333333333333
  },
  "confounding": {
    "policyA": {
      "low": {
        "onTime": 450,
        "total": 500
      },
      "high": {
        "onTime": 30,
        "total": 100
      }
    },
    "policyB": {
      "low": {
        "onTime": 95,
        "total": 100
      },
      "high": {
        "onTime": 175,
        "total": 500
      }
    },
    "pooledA": 0.8,
    "pooledB": 0.45,
    "lowRisk": [
      0.9,
      0.95
    ],
    "highRisk": [
      0.3,
      0.35
    ]
  },
  "model": {
    "slope": 0.2115728028360495,
    "intercept": 959.6198110447606,
    "xMin": 2200,
    "xMax": 8900,
    "slopePer1000": 211.5728028360495,
    "example": {
      "income": 5500,
      "spend": 2123.270226643033,
      "limit": 6400
    },
    "interceptOutside": true
  },
  "errors": {
    "mae": 224.8771150768383,
    "mse": 84416.8457817779,
    "rmse": 290.5457722662264,
    "residuals": [
      -173.07997728406963,
      -24.394537851279438,
      -99.70909841848925,
      -75.60229912750174,
      199.6617804470909,
      18.189939596275963,
      146.87537902906615,
      -124.59646182174856,
      -244.33238224715615,
      -31.382863239773542,
      8.8812163348191,
      -174.59062451599584,
      337.6734550585968,
      -296.641105508613,
      -38.69158650123063,
      -336.8488667848351,
      -31.32070763565025,
      461.78609165533726,
      503.47153108812745,
      -63.000309762687266,
      184.68512967010247,
      124.37056910289266,
      488.47736839388017,
      316.16280782667036,
      332.69096697585564,
      -568.6235935913542,
      21.06184584143557,
      229.16864513242353,
      -717.724555576589,
      -372.617756285601
    ],
    "largest": -717.724555576589
  },
  "fit": {
    "r2": 0.6815075189759174,
    "adjusted": 0.6701327875107717,
    "sse": 2532505.373453337,
    "sst": 7951538.966666667,
    "withDay": {
      "r2": 0.6885774625921738,
      "adjusted": 0.6655091264878904
    }
  },
  "diagnostics": {
    "lowRms": 166.98543760428234,
    "highRms": 375.43249085776006,
    "ratio": 2.2482947989000697,
    "annualR2Between": 1.0000000000000004,
    "annualIdentified": false,
    "annualRelativeDet": -4.85213790726814e-16
  },
  "extrapolation": {
    "highEarners": [
      {
        "income": 15000,
        "spend": 2188,
        "predicted": 4133.211853585503,
        "limit": 12400,
        "limitFromActual": 6600,
        "outside": true
      },
      {
        "income": 22000,
        "spend": 4352,
        "predicted": 5614.221473437849,
        "limit": 16800,
        "limitFromActual": 13100,
        "outside": true
      },
      {
        "income": 30000,
        "spend": 5511,
        "predicted": 7306.8038961262455,
        "limit": 21900,
        "limitFromActual": 16500,
        "outside": true
      }
    ],
    "fittedRange": [
      2200,
      8900
    ],
    "band": 569.4697136418037,
    "missedByBand": 3,
    "overshoot": [
      1945.2118535855034,
      1262.2214734378495,
      1795.8038961262455
    ]
  }
}

Prefer your own machine? Every file is in the course repository · open it in Codespaces.

What the demo does

A prototype suggests a starting credit limit from monthly income for 30 synthetic applicants: scatter, covariance, three correlations, a Simpson-style trap, an OLS line, residual metrics, R², diagnostics, and the extrapolation that fails. Every module 09 lesson becomes one model feature.