Module 9 · Dependence, Regression, and Model Foundations Module demo
Credit Limit Scoring Model
Income, spending and a limit.
Transcript
41 sentences · select one to jump thereCode lab
Run it yourself
The demo source in one language. Edit it, run TypeScript and Python right here, and compare with the expected output.
/**
* Fintech Math Bootcamp · Module 09 demo · Credit Limit Scoring Model
* A card issuer's prototype suggests a starting credit limit from monthly income: fit monthly card
* spending on income, then apply a policy of three months of predicted spending. How strong is the
* relationship, how good is the fit, and where does the model stop being trustworthy?
* Lessons 081–090: scatter plots, covariance, Pearson, Spearman and Kendall, confounding, simple OLS,
* interpreting coefficients, residual metrics, R² and adjusted R², heteroskedasticity and
* multicollinearity.
* Every applicant is synthetic, generated from a seeded formula. This is a teaching model, not a
* description of how any lender sets credit limits.
*/
export type Applicant = {income: number; spend: number};
// Seeded generator (mulberry32) and Box–Muller normal draws, so the data are identical on every run.
export function mulberry32(seed: number): () => number {
let a = seed >>> 0;
return () => {
a = (a + 0x6d2b79f5) >>> 0;
let t = a;
t = Math.imul(t ^ (t >>> 15), t | 1);
t ^= t + Math.imul(t ^ (t >>> 7), t | 61);
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
};
}
const normal = (rand: () => number) => Math.sqrt(-2 * Math.log(1 - rand())) * Math.cos(2 * Math.PI * rand());
// The hidden truth of this synthetic world: spending grows with income but flattens out (concave),
// and its noise widens as income rises (heteroskedastic).
export const trueSpend = (income: number) => 600 + 900 * Math.log(income / 1000);
export const noiseSd = (income: number) => 0.06 * income - 40;
const mean = (a: number[]) => a.reduce((s, v) => s + v, 0) / a.length;
const dot = (a: number[], b: number[]) => a.reduce((s, v, i) => s + v * b[i], 0);
const centered = (a: number[]) => {const m = mean(a); return a.map(v => v - m);};
// 082 · sample covariance: Σ(x − x̄)(y − ȳ) / (n − 1)
export const covariance = (x: number[], y: number[]) => dot(centered(x), centered(y)) / (x.length - 1);
// 083 · Pearson correlation: Sxy / √(Sxx Syy)
export function pearson(x: number[], y: number[]): number {
const dx = centered(x), dy = centered(y), sxx = dot(dx, dx), syy = dot(dy, dy);
if (sxx === 0 || syy === 0) throw new Error("Correlation undefined: zero variance");
return dot(dx, dy) / Math.sqrt(sxx * syy);
}
// 084 · Spearman = Pearson on ranks (ties get average ranks); Kendall tau-a = (concordant − discordant) / pairs
export const ranks = (a: number[]) => a.map(v => 1 + a.filter(z => z < v).length + (a.filter(z => z === v).length - 1) / 2);
export const spearman = (x: number[], y: number[]) => pearson(ranks(x), ranks(y));
export function kendallTauA(x: number[], y: number[]): number {
let s = 0;
for (let i = 0; i < x.length; i++) for (let j = i + 1; j < x.length; j++) s += Math.sign((x[j] - x[i]) * (y[j] - y[i]));
return s / (x.length * (x.length - 1) / 2);
}
// 085 · a pooled comparison that reverses inside every group (Simpson's paradox)
export type Cell = {onTime: number; total: number};
export function simpson(a: {low: Cell; high: Cell}, b: {low: Cell; high: Cell}) {
const r = (c: Cell) => c.onTime / c.total;
const pool = (g: {low: Cell; high: Cell}) => (g.low.onTime + g.high.onTime) / (g.low.total + g.high.total);
return {pooledA: pool(a), pooledB: pool(b), lowRisk: [r(a.low), r(b.low)], highRisk: [r(a.high), r(b.high)]};
}
// 086 · simple OLS: b1 = Sxy / Sxx, b0 = ȳ − b1 x̄
export function ols(x: number[], y: number[]) {
const dx = centered(x), sxx = dot(dx, dx);
if (sxx === 0) throw new Error("Slope undefined: no spread in x");
const slope = dot(dx, centered(y)) / sxx, intercept = mean(y) - slope * mean(x);
return {slope, intercept, xMin: Math.min(...x), xMax: Math.max(...x)};
}
// 087 · prediction = intercept + slope · input, flagged when the input is outside the fitted range
export const predict = (m: {slope: number; intercept: number}, x: number) => m.intercept + m.slope * x;
export const outsideFit = (m: {xMin: number; xMax: number}, x: number) => x < m.xMin || x > m.xMax;
// policy used by this prototype: a starting limit of three months of predicted spending, rounded to 100
export const suggestedLimit = (predictedSpend: number) => Math.round(3 * predictedSpend / 100) * 100;
// 088 · residuals and error metrics
export function errorMetrics(actual: number[], predicted: number[]) {
const e = actual.map((v, i) => v - predicted[i]);
const mse = mean(e.map(v => v * v));
return {residuals: e, mae: mean(e.map(Math.abs)), mse, rmse: Math.sqrt(mse)};
}
// 089 · R² against the mean-only baseline, and adjusted R² for p predictors
export function rSquared(actual: number[], predicted: number[], p: number) {
const sse = dot(actual.map((v, i) => v - predicted[i]), actual.map((v, i) => v - predicted[i]));
const dy = centered(actual), sst = dot(dy, dy), n = actual.length;
if (sst === 0) throw new Error("R-squared undefined: constant target");
if (n - p - 1 <= 0) throw new Error("Adjusted R-squared needs n > p+1");
const r2 = 1 - sse / sst;
return {sse, sst, r2, adjusted: 1 - (1 - r2) * (n - 1) / (n - p - 1)};
}
// two-predictor OLS through the centered normal equations (used for the extra-predictor comparison)
export function ols2(x1: number[], x2: number[], y: number[]) {
const a = centered(x1), b = centered(x2), c = centered(y);
const s11 = dot(a, a), s22 = dot(b, b), s12 = dot(a, b), s1y = dot(a, c), s2y = dot(b, c);
const det = s11 * s22 - s12 * s12;
const scale = s11 * s22; // determinant relative to its largest possible size
if (Math.abs(det) <= 1e-12 * scale) return {identified: false as const, relativeDet: det / scale};
const b1 = (s22 * s1y - s12 * s2y) / det, b2 = (s11 * s2y - s12 * s1y) / det;
const b0 = mean(y) - b1 * mean(x1) - b2 * mean(x2);
return {identified: true as const, b0, b1, b2, relativeDet: det / scale, predictions: x1.map((v, i) => b0 + b1 * v + b2 * x2[i])};
}
// 090 · heteroskedasticity: mean squared residual in the lower and upper halves of income
export function spreadByHalf(x: number[], residuals: number[]) {
const order = x.map((v, i) => [v, i]).sort((p, q) => p[0] - q[0]).map(p => p[1]);
const half = Math.floor(order.length / 2), ms = (idx: number[]) => mean(idx.map(i => residuals[i] ** 2));
return {lowRms: Math.sqrt(ms(order.slice(0, half))), highRms: Math.sqrt(ms(order.slice(half)))};
}
export function runDemo() {
const rand = mulberry32(916);
const n = 30;
// incomes spread evenly across 2,000–9,000 a month with a little jitter, rounded to 50
const applicants: Applicant[] = Array.from({length: n}, (_, i) => {
const income = Math.round((2000 + 7000 * (i + 0.5) / n + 150 * (rand() - 0.5)) / 50) * 50;
return {income, spend: Math.round(trueSpend(income) + noiseSd(income) * normal(rand))};
});
const x = applicants.map(a => a.income), y = applicants.map(a => a.spend);
// association
const cov = covariance(x, y), r = pearson(x, y), rho = spearman(x, y), tau = kendallTauA(x, y);
const dx = centered(x), dy = centered(y);
const quadrants = {agree: dx.filter((v, i) => v * dy[i] > 0).length, disagree: dx.filter((v, i) => v * dy[i] < 0).length};
// the model
const model = ols(x, y), fitted = x.map(v => predict(model, v));
const metrics = errorMetrics(y, fitted), fit = rSquared(y, fitted, 1);
const example = {income: 5500, spend: predict(model, 5500), limit: suggestedLimit(predict(model, 5500))};
// an extra predictor with no real information: the day of the month the application arrived
const day = applicants.map(() => 1 + Math.floor(rand() * 28));
const withDay = ols2(x, day, y);
const fitWithDay = withDay.identified ? rSquared(y, withDay.predictions, 2) : null;
// a redundant predictor: annual income is exactly 12 × monthly income
const annual = x.map(v => 12 * v);
const redundant = ols2(x, annual, y);
const r2Between = pearson(x, annual) ** 2;
// heteroskedasticity
const spread = spreadByHalf(x, metrics.residuals);
// extrapolation: three high earners far outside the fitted range, drawn from the same hidden truth
const highEarners = [15000, 22000, 30000].map(income => {
const spend = Math.round(trueSpend(income) + noiseSd(income) * normal(rand));
const pred = predict(model, income);
return {income, spend, predicted: pred, limit: suggestedLimit(pred), limitFromActual: suggestedLimit(spend), outside: outsideFit(model, income)};
});
// the in-sample error band the model would quote: ±1.96 × RMSE (it ignores extrapolation entirely)
const band = 1.96 * metrics.rmse;
const missedByBand = highEarners.filter(h => Math.abs(h.spend - h.predicted) > band).length;
// confounding: on-time repayment under two limit policies, split by risk band
const policyA = {low: {onTime: 450, total: 500}, high: {onTime: 30, total: 100}};
const policyB = {low: {onTime: 95, total: 100}, high: {onTime: 175, total: 500}};
return {
applicants, n,
association: {covariance: cov, pearson: r, spearman: rho, kendall: tau, quadrants, meanIncome: mean(x), meanSpend: mean(y)},
confounding: {policyA, policyB, ...simpson(policyA, policyB)},
model: {...model, slopePer1000: model.slope * 1000, example, interceptOutside: outsideFit(model, 0)},
errors: {mae: metrics.mae, mse: metrics.mse, rmse: metrics.rmse, residuals: metrics.residuals,
largest: metrics.residuals.reduce((m, e) => (Math.abs(e) > Math.abs(m) ? e : m), 0)},
fit: {r2: fit.r2, adjusted: fit.adjusted, sse: fit.sse, sst: fit.sst,
withDay: fitWithDay ? {r2: fitWithDay.r2, adjusted: fitWithDay.adjusted} : null},
diagnostics: {lowRms: spread.lowRms, highRms: spread.highRms, ratio: spread.highRms / spread.lowRms,
annualR2Between: r2Between, annualIdentified: redundant.identified, annualRelativeDet: redundant.relativeDet},
extrapolation: {highEarners, fittedRange: [model.xMin, model.xMax], band, missedByBand, overshoot: highEarners.map(h => h.predicted - h.spend)},
};
}
export const checkedResult = {"applicants":[{"income":2200,"spend":1252},{"income":2400,"spend":1443},{"income":2600,"spend":1410},{"income":2850,"spend":1487},{"income":3000,"spend":1794},{"income":3300,"spend":1676},{"income":3500,"spend":1847},{"income":3800,"spend":1639},{"income":3950,"spend":1551},{"income":4300,"spend":1838},{"income":4450,"spend":1910},{"income":4750,"spend":1790},{"income":4900,"spend":2334},{"income":5100,"spend":1742},{"income":5450,"spend":2074},{"income":5550,"spend":1797},{"income":5850,"spend":2166},{"income":6100,"spend":2712},{"income":6300,"spend":2796},{"income":6600,"spend":2293},{"income":6800,"spend":2583},{"income":7000,"spend":2565},{"income":7250,"spend":2982},{"income":7450,"spend":2852},{"income":7750,"spend":2932},{"income":7950,"spend":2073},{"income":8150,"spend":2705},{"income":8400,"spend":2966},{"income":8650,"spend":2072},{"income":8900,"spend":2470}],"n":30,"association":{"covariance":883210.0574712645,"pearson":0.8255346867188061,"spearman":0.8269187986651836,"kendall":0.6459770114942529,"quadrants":{"agree":26,"disagree":4},"meanIncome":5508.333333333333,"meanSpend":2125.0333333333333},"confounding":{"policyA":{"low":{"onTime":450,"total":500},"high":{"onTime":30,"total":100}},"policyB":{"low":{"onTime":95,"total":100},"high":{"onTime":175,"total":500}},"pooledA":0.8,"pooledB":0.45,"lowRisk":[0.9,0.95],"highRisk":[0.3,0.35]},"model":{"slope":0.2115728028360495,"intercept":959.6198110447606,"xMin":2200,"xMax":8900,"slopePer1000":211.5728028360495,"example":{"income":5500,"spend":2123.270226643033,"limit":6400},"interceptOutside":true},"errors":{"mae":224.8771150768383,"mse":84416.8457817779,"rmse":290.5457722662264,"residuals":[-173.07997728406963,-24.394537851279438,-99.70909841848925,-75.60229912750174,199.6617804470909,18.189939596275963,146.87537902906615,-124.59646182174856,-244.33238224715615,-31.382863239773542,8.8812163348191,-174.59062451599584,337.6734550585968,-296.641105508613,-38.69158650123063,-336.8488667848351,-31.32070763565025,461.78609165533726,503.47153108812745,-63.000309762687266,184.68512967010247,124.37056910289266,488.47736839388017,316.16280782667036,332.69096697585564,-568.6235935913542,21.06184584143557,229.16864513242353,-717.724555576589,-372.617756285601],"largest":-717.724555576589},"fit":{"r2":0.6815075189759174,"adjusted":0.6701327875107717,"sse":2532505.373453337,"sst":7951538.966666667,"withDay":{"r2":0.6885774625921738,"adjusted":0.6655091264878904}},"diagnostics":{"lowRms":166.98543760428234,"highRms":375.43249085776006,"ratio":2.2482947989000697,"annualR2Between":1.0000000000000004,"annualIdentified":false,"annualRelativeDet":-4.85213790726814e-16},"extrapolation":{"highEarners":[{"income":15000,"spend":2188,"predicted":4133.211853585503,"limit":12400,"limitFromActual":6600,"outside":true},{"income":22000,"spend":4352,"predicted":5614.221473437849,"limit":16800,"limitFromActual":13100,"outside":true},{"income":30000,"spend":5511,"predicted":7306.8038961262455,"limit":21900,"limitFromActual":16500,"outside":true}],"fittedRange":[2200,8900],"band":569.4697136418037,"missedByBand":3,"overshoot":[1945.2118535855034,1262.2214734378495,1795.8038961262455]}};
// Run this file directly: npx tsx lessons/09-dependence-regression-and-model-foundations/demo-credit-limit-scoring-model.ts
if (process.argv[1] && import.meta.url.endsWith(process.argv[1].replace(/\\/g, "/").split("/").pop()!)) {
console.log(JSON.stringify(runDemo(), null, 2));
}
Your output
Press Run to execute the code in your browser.
Expected output
{
"applicants": [
{
"income": 2200,
"spend": 1252
},
{
"income": 2400,
"spend": 1443
},
{
"income": 2600,
"spend": 1410
},
{
"income": 2850,
"spend": 1487
},
{
"income": 3000,
"spend": 1794
},
{
"income": 3300,
"spend": 1676
},
{
"income": 3500,
"spend": 1847
},
{
"income": 3800,
"spend": 1639
},
{
"income": 3950,
"spend": 1551
},
{
"income": 4300,
"spend": 1838
},
{
"income": 4450,
"spend": 1910
},
{
"income": 4750,
"spend": 1790
},
{
"income": 4900,
"spend": 2334
},
{
"income": 5100,
"spend": 1742
},
{
"income": 5450,
"spend": 2074
},
{
"income": 5550,
"spend": 1797
},
{
"income": 5850,
"spend": 2166
},
{
"income": 6100,
"spend": 2712
},
{
"income": 6300,
"spend": 2796
},
{
"income": 6600,
"spend": 2293
},
{
"income": 6800,
"spend": 2583
},
{
"income": 7000,
"spend": 2565
},
{
"income": 7250,
"spend": 2982
},
{
"income": 7450,
"spend": 2852
},
{
"income": 7750,
"spend": 2932
},
{
"income": 7950,
"spend": 2073
},
{
"income": 8150,
"spend": 2705
},
{
"income": 8400,
"spend": 2966
},
{
"income": 8650,
"spend": 2072
},
{
"income": 8900,
"spend": 2470
}
],
"n": 30,
"association": {
"covariance": 883210.0574712645,
"pearson": 0.8255346867188061,
"spearman": 0.8269187986651836,
"kendall": 0.6459770114942529,
"quadrants": {
"agree": 26,
"disagree": 4
},
"meanIncome": 5508.333333333333,
"meanSpend": 2125.0333333333333
},
"confounding": {
"policyA": {
"low": {
"onTime": 450,
"total": 500
},
"high": {
"onTime": 30,
"total": 100
}
},
"policyB": {
"low": {
"onTime": 95,
"total": 100
},
"high": {
"onTime": 175,
"total": 500
}
},
"pooledA": 0.8,
"pooledB": 0.45,
"lowRisk": [
0.9,
0.95
],
"highRisk": [
0.3,
0.35
]
},
"model": {
"slope": 0.2115728028360495,
"intercept": 959.6198110447606,
"xMin": 2200,
"xMax": 8900,
"slopePer1000": 211.5728028360495,
"example": {
"income": 5500,
"spend": 2123.270226643033,
"limit": 6400
},
"interceptOutside": true
},
"errors": {
"mae": 224.8771150768383,
"mse": 84416.8457817779,
"rmse": 290.5457722662264,
"residuals": [
-173.07997728406963,
-24.394537851279438,
-99.70909841848925,
-75.60229912750174,
199.6617804470909,
18.189939596275963,
146.87537902906615,
-124.59646182174856,
-244.33238224715615,
-31.382863239773542,
8.8812163348191,
-174.59062451599584,
337.6734550585968,
-296.641105508613,
-38.69158650123063,
-336.8488667848351,
-31.32070763565025,
461.78609165533726,
503.47153108812745,
-63.000309762687266,
184.68512967010247,
124.37056910289266,
488.47736839388017,
316.16280782667036,
332.69096697585564,
-568.6235935913542,
21.06184584143557,
229.16864513242353,
-717.724555576589,
-372.617756285601
],
"largest": -717.724555576589
},
"fit": {
"r2": 0.6815075189759174,
"adjusted": 0.6701327875107717,
"sse": 2532505.373453337,
"sst": 7951538.966666667,
"withDay": {
"r2": 0.6885774625921738,
"adjusted": 0.6655091264878904
}
},
"diagnostics": {
"lowRms": 166.98543760428234,
"highRms": 375.43249085776006,
"ratio": 2.2482947989000697,
"annualR2Between": 1.0000000000000004,
"annualIdentified": false,
"annualRelativeDet": -4.85213790726814e-16
},
"extrapolation": {
"highEarners": [
{
"income": 15000,
"spend": 2188,
"predicted": 4133.211853585503,
"limit": 12400,
"limitFromActual": 6600,
"outside": true
},
{
"income": 22000,
"spend": 4352,
"predicted": 5614.221473437849,
"limit": 16800,
"limitFromActual": 13100,
"outside": true
},
{
"income": 30000,
"spend": 5511,
"predicted": 7306.8038961262455,
"limit": 21900,
"limitFromActual": 16500,
"outside": true
}
],
"fittedRange": [
2200,
8900
],
"band": 569.4697136418037,
"missedByBand": 3,
"overshoot": [
1945.2118535855034,
1262.2214734378495,
1795.8038961262455
]
}
}Prefer your own machine? Every file is in the course repository · open it in Codespaces.
What the demo does
A prototype suggests a starting credit limit from monthly income for 30 synthetic applicants: scatter, covariance, three correlations, a Simpson-style trap, an OLS line, residual metrics, R², diagnostics, and the extrapolation that fails. Every module 09 lesson becomes one model feature.
Lessons it combines
- Scatter Plots, Association, and Nonlinear Patterns
- Sample Covariance Calculation and Interpretation
- Pearson Correlation Calculation and Interpretation
- Spearman Rank Correlation and Kendall Tau
- Correlation, Causation, Confounding, and Spurious Relationships
- Simple Ordinary Least Squares Regression
- Intercepts, Slopes, Coefficients, and Predictions
- Residuals, MAE, MSE, and RMSE
- R-Squared and Adjusted R-Squared
- Regression Assumptions, Heteroskedasticity, and Multicollinearity