diff --git a/packages/cli/src/music-render.ts b/packages/cli/src/music-render.ts index ffccea7..d16faf1 100644 --- a/packages/cli/src/music-render.ts +++ b/packages/cli/src/music-render.ts @@ -182,7 +182,8 @@ export async function renderScore( const decoded = spawnSync( resolveFfmpeg(), ["-v", "error", "-i", stem, "-f", "f32le", "-ar", "48000", "-ac", "2", "pipe:1"], - { maxBuffer: 256 * 1024 * 1024 }, + // f32 stereo at 48 kHz, plus headroom: long scores exceed a fixed cap. + { maxBuffer: Math.max(256 * 1024 * 1024, Math.ceil((compiled.durationSeconds + 2) * 48000 * 8)) }, ); if (decoded.status !== 0) throw new Error(`Cannot decode ${t.id} for room processing`); const n = decoded.stdout.length / 8, diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 280ed74..64320fd 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -162,6 +162,31 @@ importers: specifier: ^19.0.0 version: 19.2.5(@types/react@19.2.18) + videos/auc-ml: + dependencies: + '@archastro/clapper': + specifier: workspace:* + version: link:../../packages/cli + '@archastro/clapper-core': + specifier: workspace:* + version: link:../../packages/core + '@archastro/clapper-music': + specifier: workspace:* + version: link:../../packages/music + react: + specifier: ^19.2.0 + version: 19.2.8 + react-dom: + specifier: ^19.2.0 + version: 19.2.8(react@19.2.8) + devDependencies: + '@types/react': + specifier: ^19.0.0 + version: 19.2.18 + '@types/react-dom': + specifier: ^19.0.0 + version: 19.2.5(@types/react@19.2.18) + videos/cat-ballet: dependencies: '@archastro/clapper': diff --git a/third-party/fonts/manifest.json b/third-party/fonts/manifest.json index 8272fcc..c28f691 100644 --- a/third-party/fonts/manifest.json +++ b/third-party/fonts/manifest.json @@ -66,6 +66,18 @@ { "file": "videos/intern-promo/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB4NHhqcL71Q.woff2", "sha256": "4f4dc27f4a770c0d02fde800daa836c8adc0d1e423b28da74baaf0d1cc3ab96c" + }, + { + "file": "videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB41HhqcL71QxtQ.woff2", + "sha256": "e3085219252209a8d4128f90ff6d793315c752f73a979082d1d85d9cb46f63a4" + }, + { + "file": "videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB45HhqcL71QxtQ.woff2", + "sha256": "ed108107e74221cc5e163f12c6b35ba3e75e7615283828d6862c7a0784e1b0c9" + }, + { + "file": "videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB4NHhqcL71Q.woff2", + "sha256": "4f4dc27f4a770c0d02fde800daa836c8adc0d1e423b28da74baaf0d1cc3ab96c" } ] }, @@ -90,6 +102,22 @@ { "file": "videos/intern-promo/src/fonts/jizBRFtNs2ka5fXjeivQ4LroWlx-6zUTjnTLgNs.woff2", "sha256": "60c06664b5a95c7de6cc3e00d1f9034d78bd1e40b564016b241674449a067d4d" + }, + { + "file": "videos/auc-ml/src/fonts/jizBRFtNs2ka5fXjeivQ4LroWlx-6zUTjnTLgNs.woff2", + "sha256": "60c06664b5a95c7de6cc3e00d1f9034d78bd1e40b564016b241674449a067d4d" + }, + { + "file": "videos/auc-ml/src/fonts/jizBRFtNs2ka5fXjeivQ4LroWlx-6zsTjnTLgNuZ5w.woff2", + "sha256": "a8c4bd7cd7073180e740d2d83a616b5cb0845579b73207eeafeae8532e70c901" + }, + { + "file": "videos/auc-ml/src/fonts/jizHRFtNs2ka5fXjeivQ4LroWlx-6zAjgn7Motmp5r61.woff2", + "sha256": "a04fc7ed18a8037149ce0bfda58076709d8e0840e136ed00abbdc196b7992443" + }, + { + "file": "videos/auc-ml/src/fonts/jizHRFtNs2ka5fXjeivQ4LroWlx-6zAjjH7Motmp5g.woff2", + "sha256": "6ee678c33f388dd7ba59700ebea635deb98821baafd817b09891f7927177f702" } ] }, @@ -106,6 +134,14 @@ { "file": "videos/intern-promo/src/fonts/Jqz55SSPQuCQF3t8uOwiUL-taUTtap9GayojdSFO.woff2", "sha256": "e3b56e90510a84ac0ed465b822e112983eaf58e37436bf769681c31f77b1f3a7" + }, + { + "file": "videos/auc-ml/src/fonts/Jqz55SSPQuCQF3t8uOwiUL-taUTtap9GayojdSFO.woff2", + "sha256": "e3b56e90510a84ac0ed465b822e112983eaf58e37436bf769681c31f77b1f3a7" + }, + { + "file": "videos/auc-ml/src/fonts/Jqz55SSPQuCQF3t8uOwiUL-taUTtap9IayojdSFOd1I.woff2", + "sha256": "f5df746dbc2eccec6d22fdaaa2d253cd6915a30ce9f6ba86138bb6151c7254db" } ] }, diff --git a/videos/auc-ml/README.md b/videos/auc-ml/README.md new file mode 100644 index 0000000..3e52b73 --- /dev/null +++ b/videos/auc-ml/README.md @@ -0,0 +1,56 @@ +# AUC in machine learning + +An intuition-first narrated explanation, 12:23.5 at 1920×1080 / 30 fps. One synthetic +six-payment example connects thresholds, ROC geometry, pairwise ranking and average +precision, then returns for the closing comparison (7/9 vs 29/36). A short survey covers +partial and multiclass AUC and other curves that share the letters. + +- Composition: `auc-ml`; entry: `src/index.tsx`. Title card after the cold-open hook; end card closes. +- Narration: Kokoro `af_heart`, speed 1, locked in `clapper-voices.lock.json`. One cue **per sentence**, + placed from measured takes, so each diagram step lands on the sentence that explains it + (`lineStarts()` in `src/plan.ts`). +- Score: `src/score.ts`, a quiet 60 BPM piano/viola/cello bed with harp glints at part changes. + It dips under speech by automation derived from the same timing. +- Type: Schibsted Grotesk / Fragment Mono / Instrument Serif (`src/fonts`, full Latin coverage). +- [Research and conventions](SOURCES.md) · [full transcript](transcript.md). +- `python3 videos/auc-ml/verify_math.py` checks every on-screen number with exact fractions. + +## Retiming after a script edit + +```fish +node prepare-timing.mjs probe # src/probe.json, one cue per sentence +pnpm exec clapper narration render src/probe.json -o out/probe-lines +node prepare-timing.mjs # src/timing.json, chapters, transcript +``` + +Unchanged sentences reuse cached takes. Scenes, narration and score all read `src/timing.json`. + +- Preview: `pnpm --dir videos/auc-ml preview`. +- Render: `pnpm --dir videos/auc-ml render` (adds native MP4 chapter metadata → `out/auc-ml.mp4`). + +## Chapters + +| Start | Chapter | +| --- | --- | +| 00:00 | Who should be reviewed first? | +| 00:34 | Scores → thresholds → curve → area | +| 01:05 | A score gives an ordering | +| 01:36 | One threshold gives one decision | +| 02:06 | Two rates, two different denominators | +| 02:38 | Lower the threshold. Trace the tradeoff. | +| 03:05 | Width × height, added across the curve | +| 03:37 | Area is also a ranking game | +| 04:06 | Each strip is one column of comparisons | +| 04:36 | The probability behind ROC-AUC | +| 05:08 | Equal scores mean no ranking preference | +| 05:40 | 1 is perfect. 0.5 is the chance reference. | +| 06:10 | Same ordering. Same AUC. Different numbers. | +| 06:40 | A small false-positive rate can mean many alarms | +| 07:21 | Precision asks about the flagged pile | +| 07:58 | Reward precision when a new positive is found | +| 08:30 | PR-AUC and AP are not interchangeable labels | +| 09:09 | Partial AUC: only the region you operate in | +| 09:40 | Multiclass AUC needs a decomposition and an average | +| 10:26 | Same letters, different curves | +| 11:14 | Keep the curve, the data, and the decision separate | +| 11:43 | Ask five questions when someone says “AUC” | diff --git a/videos/auc-ml/SOURCES.md b/videos/auc-ml/SOURCES.md new file mode 100644 index 0000000..49469e4 --- /dev/null +++ b/videos/auc-ml/SOURCES.md @@ -0,0 +1,37 @@ +# AUC in machine learning — source and scope + +A narrated, intuition-first explanation for a software engineer comfortable with +basic algebra. Core ROC and AP derivations use one original six-payment example. +Other ML contexts are explicitly new examples or a map of variants, not claims +that every use of “AUC” shares one definition. No pharmacokinetic content. + +Primary/official references checked 2026-09-15: + +1. [Fawcett, An introduction to ROC analysis (2006)](https://www.math.ucdavis.edu/~saito/data/roc/fawcett-roc.pdf): threshold sweeps, ROC geometry, ranking interpretation, chance reference and deployment cautions. +2. [scikit-learn ROC-AUC API](https://scikit-learn.org/stable/modules/generated/sklearn.metrics.roc_auc_score.html): score inputs, OvR/OvO, macro/weighted aggregation and standardized partial AUC. +3. [scikit-learn average precision API](https://scikit-learn.org/stable/modules/generated/sklearn.metrics.average_precision_score.html): recall-weighted non-interpolated AP and distinction from linear trapezoidal PR area. +4. [scikit-learn metric guide](https://scikit-learn.org/stable/modules/model_evaluation.html): ROC/PR definitions, averaging and score-based evaluation. +5. [Davis & Goadrich, The Relationship Between Precision-Recall and ROC Curves (2006)](https://research.cs.wisc.edu/techreports/2006/TR1551.pdf): relationships and interpolation caveats. The original worked numbers below are independently calculated. +6. [COCO official evaluator](https://github.com/cocodataset/cocoapi/blob/master/PythonAPI/pycocotools/cocoeval.py): detection matching, interpolated precision, 101 recall thresholds and IoU thresholds .50:.05:.95. +7. [scikit-survival cumulative/dynamic AUC](https://scikit-survival.readthedocs.io/en/stable/api/generated/sksurv.metrics.cumulative_dynamic_auc.html): event-by-horizon cases, event-free controls, censoring-aware weights and time aggregation. Use the standard FPR/TPR ROC axes from ref1; the API page's background sentence about specificity is not our axis definition. +8. [scikit-learn ROC cross-validation example](https://scikit-learn.org/stable/auto_examples/model_selection/plot_roc_crossval.html): out-of-sample evaluation and variation across splits. + +9. [TensorFlow AUC metric](https://www.tensorflow.org/api_docs/python/tf/keras/metrics/AUC): fixed-threshold discretization can approximate ROC/PR area; exact rank invariance does not promise invariant coarse-grid estimates. + +## Original numerical examples + +- Sorted scores: A+ .9, B− .8, C+ .7, D+ .6, E− .4, F− .1. +- At cutoff .65: TP2, FP1, FN1, TN2; TPR2/3, FPR1/3, precision2/3. +- ROC points: (0,0),(0,1/3),(1/3,1/3),(1/3,2/3),(1/3,1),(2/3,1),(1,1). +- ROC area: 7/9; positive-negative wins: 7/9, no ties. +- Non-interpolated AP: (1 + 2/3 + 3/4)/3 = 29/36. +- Linear trapezoidal area of raw PR points, with endpoint (recall0,precision1): 55/72. +- Partial raw ROC area over FPR[0,.1]: 1/30; maximum raw area .1. McClish-standardized (scikit-learn `max_fpr=0.1`): ½(1+(1/30−.005)/(.1−.005)) ≈ .649, hand-computed. +- Prevalence example: fixed TPR.8/FPR.1 gives precision80/90 for100P/100N, but8/107 for10P/990N. Invariance is conditional on keeping class-conditional scoring behavior fixed. +- Interpolated precision envelope (max precision at recall ≥ r): 1 to recall 1/3, then 3/4. +- Retrieval callback: relevant results at ranks 1, 3, 4 give the same AP, 29/36. +- Separate multiclass illustration (labelled as three fraud types): per-class areas .95,.80,.60, supports80,15,5. Macro47/60≈.78333; support-weighted.91. +- Detection illustration: equal100×100 boxes offset20 horizontally have IoU2/3. Matching passes a.5 IoU cutoff and fails.75. + +`verify_math.py` checks these independently with exact rational arithmetic. +The six-item sample teaches mechanics; it cannot establish real model quality. diff --git a/videos/auc-ml/clapper-voices.lock.json b/videos/auc-ml/clapper-voices.lock.json new file mode 100644 index 0000000..cad01eb --- /dev/null +++ b/videos/auc-ml/clapper-voices.lock.json @@ -0,0 +1,16 @@ +{ + "schema": 1, + "engine": { + "model": "kokoro-82m-v1-q8", + "revision": "1939ad2a8e416c0acfeecc08a694d14ef25f2231", + "runtime": "478f7432db253cbd8e530daca9641885f62e7c529a6a9e1ab528564aee39bc0b", + "renderer": 1 + }, + "narrators": { + "guide": { + "voice": "af_heart", + "speed": 1, + "embedding": "d583ccff3cdca2f7fae535cb998ac07e9fcb90f09737b9a41fa2734ec44a8f0b" + } + } +} diff --git a/videos/auc-ml/clapper.json b/videos/auc-ml/clapper.json new file mode 100644 index 0000000..3223109 --- /dev/null +++ b/videos/auc-ml/clapper.json @@ -0,0 +1,7 @@ +{ + "runtime": "0.4.0", + "entry": "src/index.tsx", + "composition": "auc-ml", + "narration": "src/narration.ts", + "score": "src/score.ts" +} diff --git a/videos/auc-ml/finalize.mjs b/videos/auc-ml/finalize.mjs new file mode 100644 index 0000000..c36a95d --- /dev/null +++ b/videos/auc-ml/finalize.mjs @@ -0,0 +1,47 @@ +import { spawnSync } from "node:child_process"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { resolveFfmpeg } from "@archastro/clapper"; + +const root = path.dirname(fileURLToPath(import.meta.url)); +const chapters = JSON.parse(fs.readFileSync(path.join(root, "out/chapters.json"))); +const escape = (s) => s.replace(/[\\=;#\n]/g, (c) => "\\" + c); +const lines = [";FFMETADATA1", "title=AUC in machine learning — intuition and mathematical rigor"]; +for (const c of chapters) + lines.push( + "[CHAPTER]", + "TIMEBASE=1/1000", + `START=${Math.round(c.startSeconds * 1000)}`, + `END=${Math.round((c.startSeconds + c.durationSeconds) * 1000)}`, + `title=${escape(c.title)}`, + ); +const metadata = path.join(root, "out/chapters.ffmeta"); +fs.writeFileSync(metadata, lines.join("\n") + "\n"); +const r = spawnSync( + resolveFfmpeg(), + [ + "-hide_banner", + "-loglevel", + "error", + "-i", + path.join(root, "out/render.mp4"), + "-i", + metadata, + "-map", + "0", + "-map_metadata", + "1", + "-map_chapters", + "1", + "-c", + "copy", + "-movflags", + "+faststart", + "-y", + path.join(root, "out/auc-ml.mp4"), + ], + { stdio: "inherit" }, +); +if (r.status !== 0) throw Error("Chapter mux failed"); +console.log("Chaptered MP4: out/auc-ml.mp4"); diff --git a/videos/auc-ml/package.json b/videos/auc-ml/package.json new file mode 100644 index 0000000..5c016f0 --- /dev/null +++ b/videos/auc-ml/package.json @@ -0,0 +1,22 @@ +{ + "name": "auc-ml", + "version": "0.1.0", + "private": true, + "type": "module", + "scripts": { + "preview": "clapper preview", + "render": "clapper render -c auc-ml --crf 20 -o out/render.mp4 && node finalize.mjs", + "typecheck": "tsc --noEmit" + }, + "dependencies": { + "@archastro/clapper-core": "workspace:*", + "@archastro/clapper": "workspace:*", + "react": "^19.2.0", + "react-dom": "^19.2.0", + "@archastro/clapper-music": "workspace:*" + }, + "devDependencies": { + "@types/react": "^19.0.0", + "@types/react-dom": "^19.0.0" + } +} diff --git a/videos/auc-ml/prepare-timing.mjs b/videos/auc-ml/prepare-timing.mjs new file mode 100644 index 0000000..93d1e2c --- /dev/null +++ b/videos/auc-ml/prepare-timing.mjs @@ -0,0 +1,101 @@ +// Narration timing from natural paragraph takes. +// node prepare-timing.mjs probe → src/probe.json (one cue per scene, generous windows) +// clapper narration render src/probe.json -o out/probe-para +// node prepare-timing.mjs → src/timing.json, out/chapters.json, transcript.md +// Each scene is one take (Kokoro's own sentence rhythm). Sentence onsets inside the take +// come from its pauses (sentence-bounds.mjs), so scenes can animate on the sentence that +// explains each step (lineStarts in src/plan.ts). +import assert from "node:assert/strict"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { sentenceOnsets } from "./sentence-bounds.mjs"; + +const root = path.dirname(fileURLToPath(import.meta.url)); +const beats = JSON.parse(fs.readFileSync(path.join(root, "src/beats.json"))); +export const sentences = (text) => text.split(/(?<=[.?!])\s+(?=[A-Z“"(])/).filter(Boolean); + +const LEAD = 0.6; // scene start → take start (the take carries ~0.3 s of its own lead-in) +const HOLD = { default: 1.8, math: 3.0 }; // take end → cut +const math = new Set([ + "area", + "pairs", + "bridge", + "rigor", + "ties", + "imbalance", + "ap", + "conventions", + "partial", + "multiclass", +]); + +if (process.argv[2] === "probe") { + let at = 0; + const cues = []; + for (const b of beats) { + if (!b.text) continue; + const window = Math.max(40, Math.ceil(b.text.split(/\s+/).length / 1.5) + 10); + cues.push({ id: b.id, narrator: "guide", text: b.text, at, duration: window }); + at += window; + } + const probe = { + title: "AUC in machine learning — paragraph probe", + narrators: { guide: { voice: "af_heart", speed: 1 } }, + cues, + }; + fs.writeFileSync(path.join(root, "src/probe.json"), JSON.stringify(probe, null, 2) + "\n"); + console.log(`${cues.length} paragraph cues → src/probe.json`); + process.exit(0); +} + +const measured = JSON.parse(fs.readFileSync(path.join(root, "out/probe-para/manifest.json"))); +const takes = new Map(); +const takeDir = path.join(root, ".clapper/narration/takes"); +for (const f of fs.readdirSync(takeDir).filter((f) => f.endsWith(".json"))) { + const t = JSON.parse(fs.readFileSync(path.join(takeDir, f))); + takes.set(t.sha256, t); +} +const timing = {}; +const chapters = []; +let at = 0; +for (const b of beats) { + let seconds = b.seconds, + take, + lines = []; + if (b.text) { + const c = measured.cues.find((c) => c.id === b.id); + assert.ok(c, `No measured take for ${b.id}`); + assert.equal(c.text, b.text, `Stale measured text for ${b.id}`); + const file = takes.get(c.sha256)?.file; + assert.ok(file, `Take audio for ${b.id} not in the narration cache`); + const parts = sentences(b.text); + const r = sentenceOnsets(file, parts.length); + lines = r.onsets.map((o, k) => { + const end = k + 1 < r.onsets.length ? r.onsets[k + 1] : r.voiceEnd; + // Plausibility: speech runs ~5–12 cs per character; a misplaced boundary breaks that. + const rate = ((end - o) / parts[k].length) * 100; + assert.ok( + rate > 4 && rate < 13, + `${b.id} sentence ${k}: implausible boundary (${rate.toFixed(1)} cs/char)`, + ); + return { at: Math.round((LEAD + o) * 100) / 100, dur: Math.round((end - o) * 100) / 100 }; + }); + take = { at: LEAD, dur: c.durationSeconds }; + seconds = Math.ceil((LEAD + c.durationSeconds + (math.has(b.id) ? HOLD.math : HOLD.default)) * 2) / 2; + } + timing[b.id] = { seconds, take, lines }; + if (b.title) chapters.push({ id: b.id, title: b.title, startSeconds: at, durationSeconds: seconds }); + at += seconds; +} +fs.writeFileSync(path.join(root, "src/timing.json"), JSON.stringify(timing, null, 2) + "\n"); +fs.mkdirSync(path.join(root, "out"), { recursive: true }); +fs.writeFileSync(path.join(root, "out/chapters.json"), JSON.stringify(chapters, null, 2) + "\n"); +const stamp = (s) => `${String(Math.floor(s / 60)).padStart(2, "0")}:${(s % 60).toFixed(1).padStart(4, "0")}`; +fs.writeFileSync( + path.join(root, "transcript.md"), + chapters + .map((c) => `## ${stamp(c.startSeconds)} — ${c.title}\n\n${beats.find((x) => x.id === c.id).text}\n`) + .join("\n"), +); +console.log(`Film: ${at}s (${(at / 60).toFixed(2)} min), ${chapters.length} chapters.`); diff --git a/videos/auc-ml/sentence-bounds.mjs b/videos/auc-ml/sentence-bounds.mjs new file mode 100644 index 0000000..7f47ae2 --- /dev/null +++ b/videos/auc-ml/sentence-bounds.mjs @@ -0,0 +1,39 @@ +// Sentence onsets inside one paragraph take: the N−1 longest pauses separate N sentences. +import fs from "node:fs"; +export function sentenceOnsets(file, n, rate = 24000) { + const buf = fs.readFileSync(file); + const x = new Float32Array(buf.buffer, buf.byteOffset, buf.length / 4); + const hop = rate / 100; // 10 ms frames + const env = []; + for (let i = 0; i + hop <= x.length; i += hop) { + let s = 0; + for (let j = i; j < i + hop; j++) s += x[j] * x[j]; + env.push(10 * Math.log10(s / hop + 1e-12)); + } + const peak = Math.max(...env), + floor = peak - 38; + const voiced = env.map((v) => v > floor); + const first = voiced.indexOf(true), + last = voiced.lastIndexOf(true); + const gaps = []; + for (let i = first, start = -1; i <= last; i++) { + if (!voiced[i] && start < 0) start = i; + if (voiced[i] && start >= 0) { + gaps.push({ start: start / 100, end: i / 100, len: (i - start) / 100 }); + start = -1; + } + } + const chosen = [...gaps] + .sort((a, b) => b.len - a.len) + .slice(0, n - 1) + .sort((a, b) => a.start - b.start); + if (chosen.length !== n - 1) + throw new Error(`${file}: found ${chosen.length + 1} sentences, expected ${n}`); + return { + onsets: [first / 100, ...chosen.map((g) => g.end)], + voiceEnd: last / 100, + pauses: chosen.map((g) => g.len), + weakest: chosen.length ? Math.min(...chosen.map((g) => g.len)) : 0, + nextGap: gaps.length >= n ? ([...gaps].sort((a, b) => b.len - a.len)[n - 1]?.len ?? 0) : 0, + }; +} diff --git a/videos/auc-ml/src/beats.json b/videos/auc-ml/src/beats.json new file mode 100644 index 0000000..b04bc6a --- /dev/null +++ b/videos/auc-ml/src/beats.json @@ -0,0 +1,164 @@ +[ + { + "id": "intro", + "chapter": "01 / THE IDEA", + "title": "Who should be reviewed first?", + "sub": "Six payments. One fraud model. A small review budget.", + "text": "A fraud model scores every payment. You can only review a few today, so the highest scores go first. But how do you know that ordering is any good, before you pick a cutoff? One number answers that: the area under a curve, or A U C. You'll compute it by hand from these six payments, then see why the same three letters can mean very different things." + }, + { + "id": "title", + "seconds": 7 + }, + { + "id": "map", + "chapter": "01 / THE WHOLE JOURNEY", + "title": "Scores → thresholds → curve → area", + "sub": "One complete picture before we open the pieces.", + "text": "Here is the whole journey. On held-out examples, the model produces scores. A threshold turns each score into a yes or no decision. Move the threshold, and each setting gives one point. Plot those points as a curve, then summarize the curve with its area. We'll use the same six payments for two kinds of curve, so you can see exactly what changes." + }, + { + "id": "scores", + "chapter": "02 / BUILD ROC", + "title": "A score gives an ordering", + "sub": "Synthetic held-out example: three frauds (+), three legitimate payments (−).", + "text": "Here are the six payments, sorted from highest score to lowest. Plus means actual fraud; minus means legitimate. The scores only need to put payments in order; they need not be calibrated probabilities. Notice one mistake: payment B is legitimate, yet it ranks above two frauds, C and D. Keep B in mind. Its mistake will become a missing piece of area." + }, + { + "id": "threshold", + "chapter": "02 / BUILD ROC", + "title": "One threshold gives one decision", + "sub": "Flag a payment when its score is at least 0.65.", + "text": "Set the threshold at zero point six five. We flag A, B, and C. A and C are frauds we caught: true positives. B is a legitimate payment we flagged: a false positive. D is a fraud we missed: a false negative. And E and F are correctly left alone: true negatives. One threshold gives one confusion matrix, not yet a curve." + }, + { + "id": "rates", + "chapter": "02 / BUILD ROC", + "title": "Two rates, two different denominators", + "sub": "Each rate divides by a different group of payments.", + "text": "The true positive rate asks: of all the actual frauds, what fraction did we catch? Two out of three. The false positive rate asks: of all the legitimate payments, what fraction did we wrongly flag? One out of three. An R O C curve, short for receiver operating characteristic, plots false positive rate across and true positive rate up. Our threshold becomes this single point." + }, + { + "id": "sweep", + "chapter": "02 / BUILD ROC", + "title": "Lower the threshold. Trace the tradeoff.", + "sub": "The threshold is a control knob, not an axis on this plot.", + "text": "Start above the highest score: nothing is flagged, so both rates are zero. Now lower the threshold past one payment at a time. Each fraud steps the curve up by a third; each legitimate payment steps it right by a third. At the bottom, everything is flagged, and we reach one, one. That staircase is the R O C curve." + }, + { + "id": "area", + "chapter": "02 / THE AREA", + "title": "Width × height, added across the curve", + "sub": "AUC = the area under this ROC curve, inside the unit square.", + "text": "Now shade the area under the curve. Each legitimate payment adds a vertical strip, one third wide. The heights are one third, one, and one. So the area is one ninth plus one third plus one third: seven ninths, about zero point seven eight. The empty corner above the first strip is payment B's mistake. And A U C is this area, not the accuracy at any one threshold." + }, + { + "id": "pairs", + "chapter": "03 / THE DEEPER INTUITION", + "title": "Area is also a ranking game", + "sub": "Pair every fraud with every legitimate payment.", + "text": "There's a second way to get the same answer. Pair every fraud with every legitimate payment, and give a point whenever the fraud scores higher. Three frauds times three legitimate payments makes nine pairs. A beats all three. C and D each beat two, and both lose to B. Seven wins out of nine: exactly seven ninths again." + }, + { + "id": "bridge", + "chapter": "03 / WHY THE TWO VIEWS AGREE", + "title": "Each strip is one column of comparisons", + "sub": "A strip's height = the fraction of frauds that outrank that legitimate payment.", + "text": "Why do they agree? Take legitimate payment B, scored zero point eight. Only one of the three frauds outranks it, so its strip is one third tall. Legitimate payments E and F are outranked by all three frauds, so their strips are full height. Averaging the strip heights counts exactly the winning pairs. That's the bridge from area to probability." + }, + { + "id": "rigor", + "chapter": "03 / MATHEMATICAL STATEMENT", + "title": "The probability behind ROC-AUC", + "sub": "Assume larger scores mean more likely to be fraud.", + "text": "Here is the exact statement. Pick one random fraud and one random legitimate payment. R O C area is the probability that the fraud scores higher, with ties counting half. On a finite sample, that is just the fraction of winning pairs, like our seven out of nine. And because the horizontal axis is false positive rate, that is what the area integrates over, not the threshold." + }, + { + "id": "ties", + "chapter": "03 / TIED SCORES", + "title": "Equal scores mean no ranking preference", + "sub": "Separate tiny example: one fraud and one legitimate payment, both scored 0.50.", + "text": "Ties need a rule. Take a separate tiny example: one fraud and one legitimate payment, both scored zero point five. Any threshold flags both or neither, so the R O C curve jumps straight from zero, zero to one, one. That diagonal has area one half, the same as giving the tied pair half credit. So a model that gives everything the same score has an A U C of one half." + }, + { + "id": "baselines", + "chapter": "03 / READ THE NUMBER", + "title": "1 is perfect. 0.5 is the chance reference.", + "sub": "Chance is an expectation, not a guarantee for every small random sample.", + "text": "A perfect ranking climbs to the top before moving right, and fills the whole square. Random ordering has an expected area of one half. A completely reversed ranking scores zero. Below one half can signal a flipped score direction, but don't flip a model just because of your test set. And an A U C of zero point eight does not mean eighty percent of predictions are correct." + }, + { + "id": "calibration", + "chapter": "03 / WHAT AUC LEAVES OUT", + "title": "Same ordering. Same AUC. Different numbers.", + "sub": "AUC measures ranking, not probability calibration.", + "text": "Now square every score. The numbers change, but the order doesn't, so the R O C curve and its area stay at seven ninths. A U C can't tell you which set of numbers makes better probabilities. A fixed cutoff has to move too: zero point six five becomes its square, about zero point four two. Check calibration separately, and pick the threshold from the costs of mistakes." + }, + { + "id": "imbalance", + "chapter": "04 / WHY PRECISION–RECALL EXISTS", + "title": "A small false-positive rate can mean many alarms", + "sub": "Same detector, two populations.", + "text": "Suppose a threshold catches eighty percent of frauds and flags ten percent of legitimate payments. With a hundred of each, that's eighty true alarms and ten false ones: about eighty nine percent precision. Now take ten frauds among nine hundred ninety legitimate payments. The same rates give eight true alarms and ninety nine false ones: only seven and a half percent precision. The R O C point never moved: eighty percent, ten percent. Precision did, because prevalence, the Greek letter pi, changes what the flagged pile is made of." + }, + { + "id": "pr", + "chapter": "04 / THE SECOND CURVE", + "title": "Precision asks about the flagged pile", + "sub": "Recall = TP / actual frauds. Precision = TP / everything flagged.", + "text": "Back to our six payments. Recall is just another name for the true positive rate. Precision asks a different question: of the payments we flagged, how many are really fraud? A precision recall curve plots recall across and precision up, as we sweep the same threshold. Each fraud we find raises recall; each legitimate payment we flag drags precision down. Before anything is flagged, precision is undefined, so plots usually start the curve at one." + }, + { + "id": "ap", + "chapter": "04 / AVERAGE PRECISION, WORKED", + "title": "Reward precision when a new positive is found", + "sub": "Frauds appear at ranks 1, 3 and 4.", + "text": "A common summary is average precision, or A P. Each new fraud raises recall by one third. At those moments, precision is one, then two thirds, then three quarters. Weight each by its recall step and add them up. Twenty-nine over thirty-six, about zero point eight one. Same predictions as our R O C area of seven ninths, but a different question, and a different number." + }, + { + "id": "conventions", + "chapter": "04 / NAMES THAT HIDE CHOICES", + "title": "PR-AUC and AP are not interchangeable labels", + "sub": "Same points, different conventions.", + "text": "Average precision sums recall steps times precision: zero point eight oh six. Join the raw precision recall points with straight lines and take trapezoids, and you get zero point seven six four. Those straight lines aren't even a valid interpolation in precision recall space. Other benchmarks use an interpolated envelope: at each recall, the best precision at that recall or higher. Here it lifts the middle step from two thirds to three quarters. And on large samples, a random ranker's baseline precision is the prevalence: one half here, only because half our payments are fraud. Always name the convention." + }, + { + "id": "partial", + "chapter": "05 / BEYOND THE FULL CURVE", + "title": "Partial AUC: only the region you operate in", + "sub": "Full ROC-AUC can hide differences at very low false-positive rates.", + "text": "Say you can tolerate almost no false alarms. Partial A U C keeps only false positive rates up to zero point one. Here the raw area is one thirtieth, out of a possible one tenth. Scikit-learn's max F P R setting rescales that so one half is chance and one is perfect: about zero point six five. Always state the range, and whether it was rescaled." + }, + { + "id": "multiclass", + "chapter": "05 / MANY CLASSES", + "title": "Multiclass AUC needs a decomposition and an average", + "sub": "Illustrative: three fraud types with 80, 15 and 5 examples.", + "text": "Now suppose a second model sorts confirmed frauds into three types. One versus rest turns each class into its own binary problem, with its own R O C area; one versus one compares pairs of classes instead. The macro average weights the classes equally: about zero point seven eight. Weighting by class size gives zero point nine one, because the biggest class is the easiest. Same model, two honest numbers with different priorities. And when an example can carry several labels, a micro average pools every example label pair into one curve; that is not an average of areas." + }, + { + "id": "letters", + "chapter": "05 / OTHER CURVES CALLED AUC", + "title": "Same letters, different curves", + "sub": "Before trusting the number, read the axes and the matching rule.", + "text": "Outside classification, the same letters show up on different curves. In search, average precision rewards relevant results near the top, and mean average precision averages it over queries. With three relevant results, at ranks one, three, and four, it's our twenty-nine over thirty-six again. Object detection first matches each predicted box to a true one by their overlap, called I O U. These two boxes overlap by two thirds: a match at one half, a miss at three quarters. Only then does it build precision recall curves, so Coco's A P is not an R O C area. Survival models compute an A U C at each time horizon t. A machine that failed by t is a case, one still running at t is a control, and one last seen before t is unknown, not a negative. And area under a learning curve measures performance across a training budget, not ranking at all." + }, + { + "id": "practice", + "chapter": "06 / USE IT WITHOUT GETTING FOOLED", + "title": "Keep the curve, the data, and the decision separate", + "sub": "Scores → held-out curve → summary + uncertainty → an operating threshold.", + "text": "In practice, feed the metric scores, not hard labels, and make sure larger means more likely positive. Evaluate on held-out data, and report uncertainty that matches how you sampled. Look at the curve and at important subgroups, not just one number. Then choose a threshold from your costs and your prevalence. No kind of A U C chooses that for you." + }, + { + "id": "recap", + "chapter": "06 / THE MENTAL MODEL", + "title": "Ask five questions when someone says “AUC”", + "sub": "Which curve? Which positive? Which population? Which range? Which average?", + "text": "Same six payments, two curves, and two different numbers: seven ninths, and twenty-nine over thirty-six. Neither is wrong; they answer different questions. R O C area asks whether frauds outrank legitimate payments. Precision recall asks how clean the flagged pile stays as recall grows. So when you hear A U C, ask: which curve, which positive class, which population, which range, and which average?" + }, + { + "id": "end", + "seconds": 10 + } +] diff --git a/videos/auc-ml/src/bookends.tsx b/videos/auc-ml/src/bookends.tsx new file mode 100644 index 0000000..0faa912 --- /dev/null +++ b/videos/auc-ml/src/bookends.tsx @@ -0,0 +1,219 @@ +import { AP_STEPS, ROC } from "./data"; +import { appear, C, ease, Plot, SVG, Text } from "./visuals"; + +/** Title card and end card: the film's identity, outside the lesson chrome. */ +export function Bookends({ id, f, duration }: { id: string; f: number; duration: number }) { + const out = id === "end" ? 1 - ease((f - (duration - 30)) / 30) : 1 - ease((f - (duration - 14)) / 14); + if (id === "title") { + const draw = ease((f - 10) / 75) * 6, + word = appear(f, 34, 22), + line = appear(f, 58, 18), + sub = appear(f, 76, 18); + return ( +
+
+ A visual lesson · machine learning metrics +
+
+ AUC +
+
+ Area under the curve +
+
+ What it measures, how to compute it by hand, and when the same letters mean something else. +
+
+ + + +
+
+ ); + } + // End card. + // The two curves from the recap stay on screen across the cut as the anchor. + const a = appear(f, 4, 18), + b = appear(f, 24, 18), + c = 1, + d = appear(f, 70, 18); + return ( + // A slow push-in keeps the closing hold alive. +
{ + const s = 1.5 * (1 + 0.025 * ease(f / duration)); + return `translate(${960 - 640 * s}px, ${540 - 360 * s}px) scale(${s})`; + })(), + }} + > +
+ AUC +
+
+ is a question, not just a number. +
+
+ + + + ROC area 7/9 + + + + AP 29/36 + + +
+
+ {["Which curve?", "Which positive?", "Which population?", "Which range?", "Which average?"].map( + (q) => ( + {q} + ), + )} +
+
+ Synthetic six-payment example; illustrative numbers, not measured model performance. Narration: + synthetic voice (Kokoro, af_heart). Sources: Fawcett 2006; Davis & Goadrich 2006; scikit-learn, + COCO and scikit-survival documentation. Made with Clapper. +
+
+ ); +} diff --git a/videos/auc-ml/src/contexts.tsx b/videos/auc-ml/src/contexts.tsx new file mode 100644 index 0000000..3b56548 --- /dev/null +++ b/videos/auc-ml/src/contexts.tsx @@ -0,0 +1,768 @@ +import { AP_STEPS, PR, ROC, STATES } from "./data"; +import type { LineAt } from "./roc"; +import { Arrow, appear, C, Caption, Card, clamp, Eq, ease, In, Plot, SVG, Text } from "./visuals"; + +type SceneProps = { id: string; f: number; duration: number; L: LineAt }; + +/** A population as a dot field: positives first, flagged dots solid. */ +function Field({ + x, + y, + cols, + pos, + neg, + tp, + fp, + r, + gap, + flag, +}: { + x: number; + y: number; + cols: number; + pos: number; + neg: number; + tp: number; + fp: number; + r: number; + gap: number; + flag: number; +}) { + const dots = []; + for (let i = 0; i < pos + neg; i++) { + const isPos = i < pos, + flagged = isPos ? i < tp : i - pos < fp, + cx = x + (i % cols) * gap, + cy = y + Math.floor(i / cols) * gap, + on = flagged ? flag : 0; + dots.push( + 0.5 ? C.pos : "#9fcbbd") : on > 0.5 ? C.neg : "#ddd5c8"} + />, + ); + } + return {dots}; +} + +export function Contexts({ id, f, L }: SceneProps) { + const at = (k: number, o = 0, fr = 0) => L(k, o, fr); + const A = (k: number, o = 0, dur = 14) => appear(f, at(k, o), dur); + + if (id === "imbalance") { + const panels = [ + { + x: 70, + k: 1, + title: "100 frauds + 100 legitimate", + pos: 100, + neg: 100, + tp: 80, + fp: 10, + cols: 20, + r: 6, + gap: 17, + pct: "88.9%", + }, + { + x: 650, + k: 2, + title: "10 frauds + 990 legitimate", + pos: 10, + neg: 990, + tp: 8, + fp: 99, + cols: 50, + r: 3.4, + gap: 10.6, + pct: "7.5%", + }, + ]; + return ( + <> + + + + Same detector: catches 80% of frauds, flags 10% of legitimate + + + {panels.map((v) => { + const show = 0.22 * A(0, 0.6) + 0.78 * A(v.k, 0.2), + flag = A(v.k === 1 ? 1 : 3, v.k === 1 ? 1.8 : 0.6, 12), + res = appear(f, v.k === 1 ? at(1, 0, 0.75) : at(3, 0, 0.75), 14); // lands as the percentage is spoken + const prec = v.tp / (v.tp + v.fp); + return ( + + + {v.title} + + + + + flagged pile: + + + {v.tp} fraud + + + + {v.fp} legitimate + + + + + precision {v.pct} + + + + ); + })} + + + + ROC point unchanged + + + + + + ); + } + + if (id === "pr") { + const k = clamp((f - at(3, 1.4)) / (at(5, 0) - at(3, 1.4))) * 6, + n = Math.floor(k + 1e-6), + s = STATES[n]; + const pts = PR.slice(1); + const startConv = A(5, 0.6); + return ( + <> + + + + Flag the top n payments + + + {STATES.slice(1).map((v, i) => ( + + + + top {v.n} + + + recall {v.tp}/3 + + + precision {v.tp}/{v.n} + + + ))} + = 1} + axes={A(3, 0)} + > + + + + + plotting convention + + + + + {n + ? `recall ${(s.tpr).toFixed(2)} · precision ${(s.precision).toFixed(2)}` + : "nothing flagged yet"} + + + + A fraud found raises recall. A legitimate payment flagged lowers precision. + + + With nothing flagged, precision is undefined; the curve's start at 1 is a convention. + + + ); + } + + if (id === "ap") { + const ranks = [ + { r: "1", p: "1/1", v: 1 }, + { r: "3", p: "2/3", v: 2 / 3 }, + { r: "4", p: "3/4", v: 3 / 4 }, + ]; + return ( + <> + + + {ranks.map((v, i) => { + const p = appear(f, at(2, 0, [0.38, 0.62, 0.85][i]), 16); + return ( + + + + + +1/3 + + + + ); + })} + + + + New fraud found at rank… + + + {ranks.map((v, i) => ( + + + + rank {v.r} + + + precision {v.p} + + + × 1/3 + + + ))} + + + + Same predictions: ROC area 7/9 ≈ 0.778, AP 29/36 ≈ 0.806. Different questions. + + + ); + } + + if (id === "conventions") { + const env: [number, number][] = [ + [0, 1], + [1 / 3, 1], + [1 / 3, 3 / 4], + [1, 3 / 4], + ]; + const prPts = PR as [number, number][]; + return ( + <> + + + + Average precision (steps) + + + + `${i ? "L" : "M"}${120 + a * 400} ${66 + (1 - b) * 234}`) + .join(" ")} + stroke={C.gold} + strokeWidth="3" + strokeDasharray="8 6" + fill="none" + /> + + interpolated envelope + + + + + + + random baseline = prevalence (1/2 here) + + + + + 29/36 ≈ 0.806 + + + + + Trapezoids (straight lines) + + + + 55/72 ≈ 0.764 + + + + + ✗ not valid in PR space + + + + + + Same points, different conventions. Always name the one behind the number. + + + ); + } + + if (id === "partial") { + const X = 100, + Y = 40, + W = 480, + H = 320, + band = 0.1; + return ( + <> + + + + + + + FPR ≤ 0.1 + + + + + + + + + + raw partial area + + + 1/30 ≈ 0.033 + + + out of a possible 0.1 (the outlined band) + + + + + scikit-learn max_fpr=0.1 (McClish): + + + + + + Report the FPR range and whether the area was standardized. + + + ); + } + + if (id === "multiclass") { + const classes = [ + { c: "Card theft", auc: 0.95, support: 80, col: C.pos }, + { c: "Account takeover", auc: 0.8, support: 15, col: C.blue }, + { c: "Refund abuse", auc: 0.6, support: 5, col: C.purple }, + ]; + const macroFocus = A(2, 0) * (1 - 0.65 * A(3, 0)); + return ( + <> + + {classes.map((v, i) => ( + + + + + {`vs rest: AUC ${v.auc.toFixed(2)}`} + + + + ))} + + + macro: equal weight + + + + + weighted: by class size + + + + + + Multilabel · micro average + + + pools every (example, label) score into one curve, which is not an average of per-label areas + + + + + + + ); + } + + if (id === "letters") { + const P = [ + { + k: 1, + x: 70, + y: 20, + title: "Search & recommendations", + sub: "AP per query; mAP averages queries", + c: C.blue, + }, + { k: 3, x: 660, y: 20, title: "Object detection", sub: "match boxes by IoU, then PR", c: C.gold }, + { k: 6, x: 70, y: 218, title: "Survival / time-to-event", sub: "an AUC at each horizon t", c: C.pos }, + { k: 8, x: 660, y: 218, title: "Learning curves", sub: "performance across a budget", c: C.purple }, + ]; + const w = 550, + h = 186; + return ( + + {P.map((p) => ( + + + + + {p.title} + + + {p.sub} + + + ))} + {/* Retrieval: ranks 1, 3, 4 relevant — the AP example again. */} + + {[1, 0, 1, 1, 0, 0].map((v, i) => ( + + + + {i + 1} + + + ))} + + + + AP 29/36 + + + {/* Detection: two 100×100 boxes, offset 20 → IoU 2/3. */} + + + + + + IoU = 2/3 + + + + + match at 0.5 · not at 0.75 + + + + {/* Survival: horizon line; a censored machine is unknown. */} + + + + t + + {[ + { e: 150, ev: true, l: "case", c: C.pos }, + { e: 260, ev: false, l: "unknown", c: C.gold }, + { e: 470, ev: false, l: "control", c: C.blue }, + ].map((m, i) => ( + + + + + {m.l} + + + ))} + + {/* Learning curve: area across a budget axis. */} + + {(() => { + const x0 = 700, + y0 = 392, + ww = 300, + hh = 92; + const pts = Array.from({ length: 31 }, (_, i) => [i / 30, 0.25 + 0.7 * (1 - Math.exp(-i / 8))]); + const d = pts.map(([a, b], i) => `${i ? "L" : "M"}${x0 + a * ww} ${y0 - b * hh}`).join(" "); + return ( + <> + + + + + budget → + + + ); + })()} + + + ); + } + + if (id === "practice") { + const steps = [ + { t: "Scores", s: "not hard labels; larger = more positive", c: C.ink }, + { t: "Held-out data", s: "uncertainty that matches your sampling", c: C.blue }, + { t: "The whole curve", s: "and important subgroups", c: C.pos }, + { t: "A threshold", s: "from costs and prevalence", c: C.gold }, + ]; + return ( + <> + + {steps.map((v, i) => ( + + + + + {i < 3 && } + + ))} + + + No kind of AUC chooses the threshold for you. + + + + + ); + } + + if (id === "recap") { + const qs = ["Which curve?", "Which positive?", "Which population?", "Which range?", "Which average?"]; + return ( + + + + + 7/9 + + + + 29/36 + + + + + neither is wrong + + + + + Do frauds outrank legitimate payments? + + + + + How clean is the flagged pile as recall grows? + + + {qs.map((q, i) => { + const p = appear(f, at(4, 1.4) + i * 21, 12); + return ( + + + + {q} + + + ); + })} + + ); + } + return null; +} diff --git a/videos/auc-ml/src/data.ts b/videos/auc-ml/src/data.ts new file mode 100644 index 0000000..600a856 --- /dev/null +++ b/videos/auc-ml/src/data.ts @@ -0,0 +1,40 @@ +export const ITEMS = [ + { id: "A", score: 0.9, positive: true }, + { id: "B", score: 0.8, positive: false }, + { id: "C", score: 0.7, positive: true }, + { id: "D", score: 0.6, positive: true }, + { id: "E", score: 0.4, positive: false }, + { id: "F", score: 0.1, positive: false }, +]; +export const CUTS = [1, 0.85, 0.75, 0.65, 0.5, 0.25, 0]; +export const STATES = Array.from({ length: 7 }, (_, n) => { + const tp = ITEMS.slice(0, n).filter((x) => x.positive).length, + fp = n - tp; + return { + n, + tp, + fp, + fn: 3 - tp, + tn: 3 - fp, + tpr: tp / 3, + fpr: fp / 3, + precision: n ? tp / n : 1, + threshold: CUTS[n], + }; +}); +export const ROC = STATES.map((s) => [s.fpr, s.tpr] as [number, number]); +export const PR = STATES.map((s) => [s.tpr, s.precision] as [number, number]); +export const AP_STEPS = [ + [0, 1], + [1 / 3, 1], + [1 / 3, 2 / 3], + [2 / 3, 2 / 3], + [2 / 3, 3 / 4], + [1, 3 / 4], +] as [number, number][]; +export const POS = ITEMS.filter((x) => x.positive), + NEG = ITEMS.filter((x) => !x.positive); +export const trap = (p: [number, number][]) => + p.slice(1).reduce((a, [x, y], i) => a + ((x - p[i][0]) * (y + p[i][1])) / 2, 0); +export const AUC = trap(ROC), + AP = (1 + 2 / 3 + 3 / 4) / 3; diff --git a/videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB41HhqcL71QxtQ.woff2 b/videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB41HhqcL71QxtQ.woff2 new file mode 100644 index 0000000..920f91f Binary files /dev/null and b/videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB41HhqcL71QxtQ.woff2 differ diff --git a/videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB45HhqcL71QxtQ.woff2 b/videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB45HhqcL71QxtQ.woff2 new file mode 100644 index 0000000..e0f855a Binary files /dev/null and b/videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB45HhqcL71QxtQ.woff2 differ diff --git a/videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB4NHhqcL71Q.woff2 b/videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB4NHhqcL71Q.woff2 new file mode 100644 index 0000000..09f041a Binary files /dev/null and b/videos/auc-ml/src/fonts/4iCr6K5wfMRRjxp0DA6-2CLnB4NHhqcL71Q.woff2 differ diff --git a/videos/auc-ml/src/fonts/Jqz55SSPQuCQF3t8uOwiUL-taUTtap9GayojdSFO.woff2 b/videos/auc-ml/src/fonts/Jqz55SSPQuCQF3t8uOwiUL-taUTtap9GayojdSFO.woff2 new file mode 100644 index 0000000..dbca3fd Binary files /dev/null and b/videos/auc-ml/src/fonts/Jqz55SSPQuCQF3t8uOwiUL-taUTtap9GayojdSFO.woff2 differ diff --git a/videos/auc-ml/src/fonts/Jqz55SSPQuCQF3t8uOwiUL-taUTtap9IayojdSFOd1I.woff2 b/videos/auc-ml/src/fonts/Jqz55SSPQuCQF3t8uOwiUL-taUTtap9IayojdSFOd1I.woff2 new file mode 100644 index 0000000..42fd93c Binary files /dev/null and b/videos/auc-ml/src/fonts/Jqz55SSPQuCQF3t8uOwiUL-taUTtap9IayojdSFOd1I.woff2 differ diff --git a/videos/auc-ml/src/fonts/fonts.css b/videos/auc-ml/src/fonts/fonts.css new file mode 100644 index 0000000..4af8baa --- /dev/null +++ b/videos/auc-ml/src/fonts/fonts.css @@ -0,0 +1,149 @@ +/* cyrillic-ext */ +@font-face { + font-family: "Fragment Mono"; + font-style: normal; + font-weight: 400; + font-display: swap; + src: url(./4iCr6K5wfMRRjxp0DA6-2CLnB45HhqcL71QxtQ.woff2) format("woff2"); + unicode-range: U+0460-052F, U+1C80-1C8A, U+20B4, U+2DE0-2DFF, U+A640-A69F, U+FE2E-FE2F; +} +/* latin-ext */ +@font-face { + font-family: "Fragment Mono"; + font-style: normal; + font-weight: 400; + font-display: swap; + src: url(./4iCr6K5wfMRRjxp0DA6-2CLnB41HhqcL71QxtQ.woff2) format("woff2"); + unicode-range: + U+0100-02BA, U+02BD-02C5, U+02C7-02CC, U+02CE-02D7, U+02DD-02FF, U+0304, U+0308, U+0329, U+1D00-1DBF, U+1E00-1E9F, U+1EF2-1EFF, U+2020, U+20A0-20AB, U+20AD-20C0, U+2113, U+2C60-2C7F, U+A720-A7FF; +} +/* latin */ +@font-face { + font-family: "Fragment Mono"; + font-style: normal; + font-weight: 400; + font-display: swap; + src: url(./4iCr6K5wfMRRjxp0DA6-2CLnB4NHhqcL71Q.woff2) format("woff2"); + unicode-range: + U+0000-00FF, U+0131, U+0152-0153, U+02BB-02BC, U+02C6, U+02DA, U+02DC, U+0304, U+0308, U+0329, U+2000-206F, U+20AC, U+2122, U+2191, U+2193, U+2212, U+2215, U+FEFF, U+FFFD; +} +/* latin-ext */ +@font-face { + font-family: "Instrument Serif"; + font-style: italic; + font-weight: 400; + font-display: swap; + src: url(./jizHRFtNs2ka5fXjeivQ4LroWlx-6zAjgn7Motmp5r61.woff2) format("woff2"); + unicode-range: + U+0100-02BA, U+02BD-02C5, U+02C7-02CC, U+02CE-02D7, U+02DD-02FF, U+0304, U+0308, U+0329, U+1D00-1DBF, U+1E00-1E9F, U+1EF2-1EFF, U+2020, U+20A0-20AB, U+20AD-20C0, U+2113, U+2C60-2C7F, U+A720-A7FF; +} +/* latin */ +@font-face { + font-family: "Instrument Serif"; + font-style: italic; + font-weight: 400; + font-display: swap; + src: url(./jizHRFtNs2ka5fXjeivQ4LroWlx-6zAjjH7Motmp5g.woff2) format("woff2"); + unicode-range: + U+0000-00FF, U+0131, U+0152-0153, U+02BB-02BC, U+02C6, U+02DA, U+02DC, U+0304, U+0308, U+0329, U+2000-206F, U+20AC, U+2122, U+2191, U+2193, U+2212, U+2215, U+FEFF, U+FFFD; +} +/* latin-ext */ +@font-face { + font-family: "Instrument Serif"; + font-style: normal; + font-weight: 400; + font-display: swap; + src: url(./jizBRFtNs2ka5fXjeivQ4LroWlx-6zsTjnTLgNuZ5w.woff2) format("woff2"); + unicode-range: + U+0100-02BA, U+02BD-02C5, U+02C7-02CC, U+02CE-02D7, U+02DD-02FF, U+0304, U+0308, U+0329, U+1D00-1DBF, U+1E00-1E9F, U+1EF2-1EFF, U+2020, U+20A0-20AB, U+20AD-20C0, U+2113, U+2C60-2C7F, U+A720-A7FF; +} +/* latin */ +@font-face { + font-family: "Instrument Serif"; + font-style: normal; + font-weight: 400; + font-display: swap; + src: url(./jizBRFtNs2ka5fXjeivQ4LroWlx-6zUTjnTLgNs.woff2) format("woff2"); + unicode-range: + U+0000-00FF, U+0131, U+0152-0153, U+02BB-02BC, U+02C6, U+02DA, U+02DC, U+0304, U+0308, U+0329, U+2000-206F, U+20AC, U+2122, U+2191, U+2193, U+2212, U+2215, U+FEFF, U+FFFD; +} +/* latin-ext */ +@font-face { + font-family: "Schibsted Grotesk"; + font-style: normal; + font-weight: 400; + font-display: swap; + src: url(./Jqz55SSPQuCQF3t8uOwiUL-taUTtap9IayojdSFOd1I.woff2) format("woff2"); + unicode-range: + U+0100-02BA, U+02BD-02C5, U+02C7-02CC, U+02CE-02D7, U+02DD-02FF, U+0304, U+0308, U+0329, U+1D00-1DBF, U+1E00-1E9F, U+1EF2-1EFF, U+2020, U+20A0-20AB, U+20AD-20C0, U+2113, U+2C60-2C7F, U+A720-A7FF; +} +/* latin */ +@font-face { + font-family: "Schibsted Grotesk"; + font-style: normal; + font-weight: 400; + font-display: swap; + src: url(./Jqz55SSPQuCQF3t8uOwiUL-taUTtap9GayojdSFO.woff2) format("woff2"); + unicode-range: + U+0000-00FF, U+0131, U+0152-0153, U+02BB-02BC, U+02C6, U+02DA, U+02DC, U+0304, U+0308, U+0329, U+2000-206F, U+20AC, U+2122, U+2191, U+2193, U+2212, U+2215, U+FEFF, U+FFFD; +} +/* latin-ext */ +@font-face { + font-family: "Schibsted Grotesk"; + font-style: normal; + font-weight: 500; + font-display: swap; + src: url(./Jqz55SSPQuCQF3t8uOwiUL-taUTtap9IayojdSFOd1I.woff2) format("woff2"); + unicode-range: + U+0100-02BA, U+02BD-02C5, U+02C7-02CC, U+02CE-02D7, U+02DD-02FF, U+0304, U+0308, U+0329, U+1D00-1DBF, U+1E00-1E9F, U+1EF2-1EFF, U+2020, U+20A0-20AB, U+20AD-20C0, U+2113, U+2C60-2C7F, U+A720-A7FF; +} +/* latin */ +@font-face { + font-family: "Schibsted Grotesk"; + font-style: normal; + font-weight: 500; + font-display: swap; + src: url(./Jqz55SSPQuCQF3t8uOwiUL-taUTtap9GayojdSFO.woff2) format("woff2"); + unicode-range: + U+0000-00FF, U+0131, U+0152-0153, U+02BB-02BC, U+02C6, U+02DA, U+02DC, U+0304, U+0308, U+0329, U+2000-206F, U+20AC, U+2122, U+2191, U+2193, U+2212, U+2215, U+FEFF, U+FFFD; +} +/* latin-ext */ +@font-face { + font-family: "Schibsted Grotesk"; + font-style: normal; + font-weight: 600; + font-display: swap; + src: url(./Jqz55SSPQuCQF3t8uOwiUL-taUTtap9IayojdSFOd1I.woff2) format("woff2"); + unicode-range: + U+0100-02BA, U+02BD-02C5, U+02C7-02CC, U+02CE-02D7, U+02DD-02FF, U+0304, U+0308, U+0329, U+1D00-1DBF, U+1E00-1E9F, U+1EF2-1EFF, U+2020, U+20A0-20AB, U+20AD-20C0, U+2113, U+2C60-2C7F, U+A720-A7FF; +} +/* latin */ +@font-face { + font-family: "Schibsted Grotesk"; + font-style: normal; + font-weight: 600; + font-display: swap; + src: url(./Jqz55SSPQuCQF3t8uOwiUL-taUTtap9GayojdSFO.woff2) format("woff2"); + unicode-range: + U+0000-00FF, U+0131, U+0152-0153, U+02BB-02BC, U+02C6, U+02DA, U+02DC, U+0304, U+0308, U+0329, U+2000-206F, U+20AC, U+2122, U+2191, U+2193, U+2212, U+2215, U+FEFF, U+FFFD; +} +/* latin-ext */ +@font-face { + font-family: "Schibsted Grotesk"; + font-style: normal; + font-weight: 700; + font-display: swap; + src: url(./Jqz55SSPQuCQF3t8uOwiUL-taUTtap9IayojdSFOd1I.woff2) format("woff2"); + unicode-range: + U+0100-02BA, U+02BD-02C5, U+02C7-02CC, U+02CE-02D7, U+02DD-02FF, U+0304, U+0308, U+0329, U+1D00-1DBF, U+1E00-1E9F, U+1EF2-1EFF, U+2020, U+20A0-20AB, U+20AD-20C0, U+2113, U+2C60-2C7F, U+A720-A7FF; +} +/* latin */ +@font-face { + font-family: "Schibsted Grotesk"; + font-style: normal; + font-weight: 700; + font-display: swap; + src: url(./Jqz55SSPQuCQF3t8uOwiUL-taUTtap9GayojdSFO.woff2) format("woff2"); + unicode-range: + U+0000-00FF, U+0131, U+0152-0153, U+02BB-02BC, U+02C6, U+02DA, U+02DC, U+0304, U+0308, U+0329, U+2000-206F, U+20AC, U+2122, U+2191, U+2193, U+2212, U+2215, U+FEFF, U+FFFD; +} diff --git a/videos/auc-ml/src/fonts/jizBRFtNs2ka5fXjeivQ4LroWlx-6zUTjnTLgNs.woff2 b/videos/auc-ml/src/fonts/jizBRFtNs2ka5fXjeivQ4LroWlx-6zUTjnTLgNs.woff2 new file mode 100644 index 0000000..1765304 Binary files /dev/null and b/videos/auc-ml/src/fonts/jizBRFtNs2ka5fXjeivQ4LroWlx-6zUTjnTLgNs.woff2 differ diff --git a/videos/auc-ml/src/fonts/jizBRFtNs2ka5fXjeivQ4LroWlx-6zsTjnTLgNuZ5w.woff2 b/videos/auc-ml/src/fonts/jizBRFtNs2ka5fXjeivQ4LroWlx-6zsTjnTLgNuZ5w.woff2 new file mode 100644 index 0000000..3ee1bf9 Binary files /dev/null and b/videos/auc-ml/src/fonts/jizBRFtNs2ka5fXjeivQ4LroWlx-6zsTjnTLgNuZ5w.woff2 differ diff --git a/videos/auc-ml/src/fonts/jizHRFtNs2ka5fXjeivQ4LroWlx-6zAjgn7Motmp5r61.woff2 b/videos/auc-ml/src/fonts/jizHRFtNs2ka5fXjeivQ4LroWlx-6zAjgn7Motmp5r61.woff2 new file mode 100644 index 0000000..dd21ed1 Binary files /dev/null and b/videos/auc-ml/src/fonts/jizHRFtNs2ka5fXjeivQ4LroWlx-6zAjgn7Motmp5r61.woff2 differ diff --git a/videos/auc-ml/src/fonts/jizHRFtNs2ka5fXjeivQ4LroWlx-6zAjjH7Motmp5g.woff2 b/videos/auc-ml/src/fonts/jizHRFtNs2ka5fXjeivQ4LroWlx-6zAjjH7Motmp5g.woff2 new file mode 100644 index 0000000..ce3c417 Binary files /dev/null and b/videos/auc-ml/src/fonts/jizHRFtNs2ka5fXjeivQ4LroWlx-6zAjjH7Motmp5g.woff2 differ diff --git a/videos/auc-ml/src/index.tsx b/videos/auc-ml/src/index.tsx new file mode 100644 index 0000000..2182de7 --- /dev/null +++ b/videos/auc-ml/src/index.tsx @@ -0,0 +1,131 @@ +import { Composition, Easing, interpolate, registerRoot, Scenes, useFrame } from "@archastro/clapper-core"; +import { ScoreAudio } from "@archastro/clapper-core/music"; +import { NarrationAudio } from "@archastro/clapper-core/narration"; +import beats from "./beats.json"; +import { Bookends } from "./bookends"; +import { Contexts } from "./contexts"; +import narration from "./narration"; +import { FPS, lineStarts, SCENES } from "./plan"; +import { RocScenes } from "./roc"; +import score from "./score"; +import "./theme.css"; + +type Beat = { id: string; chapter?: string; title?: string; sub?: string; text?: string }; +const BEATS = beats as Beat[]; +const PARTS = ["Idea", "Build ROC", "Ranking math", "Precision–recall", "Other contexts", "Use it"]; +const partOf = (b: Beat) => (b.chapter ? Number(b.chapter.slice(0, 2)) - 1 : -1); +// Rail segments are proportional to each part's running time. +const PART_SPANS = PARTS.map((_, p) => { + const ids = BEATS.filter((b) => partOf(b) === p).map((b) => b.id); + const start = Math.min(...ids.map((id) => SCENES.start(id))), + end = Math.max(...ids.map((id) => SCENES.start(id) + SCENES.duration(id))); + return { start, end }; +}); + +function Rail({ index }: { index: number }) { + const f = useFrame(), + global = SCENES.start(BEATS[index].id) + f, + part = partOf(BEATS[index]); + return ( +
+ {PARTS.map((name, p) => { + const s = PART_SPANS[p], + fill = Math.max(0, Math.min(1, (global - s.start) / (s.end - s.start))); + return ( +
+
+ {name} +
+ ); + })} +
+ ); +} + +function Shot({ index }: { index: number }) { + const f = useFrame(), + b = BEATS[index], + d = SCENES.duration(b.id), + L = lineStarts(b.id); + // Soft scene transitions: the header settles in, the canvas rises; everything eases out at the cut. + const enter = interpolate(f, [0, 16], [0, 1], { easing: Easing.outCubic }), + exit = interpolate(f, [d - 10, d], [1, 0], { easing: Easing.inCubic }), + headIn = interpolate(f, [0, 18], [14, 0], { easing: Easing.outCubic }); + if (!b.chapter) return ; + return ( +
+
+ {b.chapter} +
+

52 ? 37 : 42, + opacity: enter, + transform: `translateY(${headIn}px)`, + }} + > + {b.title} +

+

+ {b.sub} +

+
+ {b.id in ROC_IDS ? ( + + ) : ( + + )} +
+ +
+ ); +} +const ROC_IDS: Record = Object.fromEntries( + [ + "intro", + "map", + "scores", + "threshold", + "rates", + "sweep", + "area", + "pairs", + "bridge", + "rigor", + "ties", + "baselines", + "calibration", + ].map((id) => [id, true]), +); +function Film() { + return ( +
+ + + + {BEATS.map((b, i) => ( + + + + ))} + +
+ ); +} +registerRoot(() => ( + +)); diff --git a/videos/auc-ml/src/narration.ts b/videos/auc-ml/src/narration.ts new file mode 100644 index 0000000..c0f0f61 --- /dev/null +++ b/videos/auc-ml/src/narration.ts @@ -0,0 +1,22 @@ +import { defineNarration } from "@archastro/clapper-core/narration/models"; +import beats from "./beats.json"; +import { FPS, LINES, SCENES } from "./plan"; + +// One natural take per scene. Sentence onsets inside each take are measured +// (prepare-timing.mjs) so diagram steps can land on the sentence that explains them. +export default defineNarration({ + title: "AUC in machine learning — from intuition to mathematics", + narrators: { guide: { voice: "af_heart", speed: 1 } }, + cues: (beats as { id: string; text?: string }[]) + .filter((b) => b.text) + .map((b) => { + const take = LINES[b.id].take!; + return { + id: b.id, + narrator: "guide", + text: b.text!, + at: SCENES.start(b.id) / FPS + take.at, + duration: take.dur + 0.3, + }; + }), +}); diff --git a/videos/auc-ml/src/plan.ts b/videos/auc-ml/src/plan.ts new file mode 100644 index 0000000..a143b68 --- /dev/null +++ b/videos/auc-ml/src/plan.ts @@ -0,0 +1,23 @@ +import { defineScenes } from "@archastro/clapper-core"; +import beats from "./beats.json"; +import timing from "./timing.json"; +export const FPS = 30; +type Timing = Record< + string, + { seconds: number; take?: { at: number; dur: number }; lines: { at: number; dur: number }[] } +>; +const T = timing as Timing; +export const SCENES = defineScenes( + Object.fromEntries(beats.map((b) => [b.id, { seconds: T[b.id]?.seconds ?? 8 }])), + { fps: FPS }, +); +/** Scene-local frame at which sentence k of a scene's narration starts (k past the end → scene end). */ +export function lineStarts(id: string): (k: number, offsetSeconds?: number, fraction?: number) => number { + const lines = T[id]?.lines ?? []; + // `fraction` reaches into the sentence (0 = its first word, 1 = its last). + return (k, offset = 0, fraction = 0) => { + const at = k < lines.length ? lines[k].at + fraction * lines[k].dur : (T[id]?.seconds ?? 8); + return Math.round((at + offset) * FPS); + }; +} +export const LINES = T; diff --git a/videos/auc-ml/src/probe.json b/videos/auc-ml/src/probe.json new file mode 100644 index 0000000..93216ba --- /dev/null +++ b/videos/auc-ml/src/probe.json @@ -0,0 +1,165 @@ +{ + "title": "AUC in machine learning — paragraph probe", + "narrators": { + "guide": { + "voice": "af_heart", + "speed": 1 + } + }, + "cues": [ + { + "id": "intro", + "narrator": "guide", + "text": "A fraud model scores every payment. You can only review a few today, so the highest scores go first. But how do you know that ordering is any good, before you pick a cutoff? One number answers that: the area under a curve, or A U C. You'll compute it by hand from these six payments, then see why the same three letters can mean very different things.", + "at": 0, + "duration": 56 + }, + { + "id": "map", + "narrator": "guide", + "text": "Here is the whole journey. On held-out examples, the model produces scores. A threshold turns each score into a yes or no decision. Move the threshold, and each setting gives one point. Plot those points as a curve, then summarize the curve with its area. We'll use the same six payments for two kinds of curve, so you can see exactly what changes.", + "at": 56, + "duration": 52 + }, + { + "id": "scores", + "narrator": "guide", + "text": "Here are the six payments, sorted from highest score to lowest. Plus means actual fraud; minus means legitimate. The scores only need to put payments in order; they need not be calibrated probabilities. Notice one mistake: payment B is legitimate, yet it ranks above two frauds, C and D. Keep B in mind. Its mistake will become a missing piece of area.", + "at": 108, + "duration": 52 + }, + { + "id": "threshold", + "narrator": "guide", + "text": "Set the threshold at zero point six five. We flag A, B, and C. A and C are frauds we caught: true positives. B is a legitimate payment we flagged: a false positive. D is a fraud we missed: a false negative. And E and F are correctly left alone: true negatives. One threshold gives one confusion matrix, not yet a curve.", + "at": 160, + "duration": 52 + }, + { + "id": "rates", + "narrator": "guide", + "text": "The true positive rate asks: of all the actual frauds, what fraction did we catch? Two out of three. The false positive rate asks: of all the legitimate payments, what fraction did we wrongly flag? One out of three. An R O C curve, short for receiver operating characteristic, plots false positive rate across and true positive rate up. Our threshold becomes this single point.", + "at": 212, + "duration": 54 + }, + { + "id": "sweep", + "narrator": "guide", + "text": "Start above the highest score: nothing is flagged, so both rates are zero. Now lower the threshold past one payment at a time. Each fraud steps the curve up by a third; each legitimate payment steps it right by a third. At the bottom, everything is flagged, and we reach one, one. That staircase is the R O C curve.", + "at": 266, + "duration": 50 + }, + { + "id": "area", + "narrator": "guide", + "text": "Now shade the area under the curve. Each legitimate payment adds a vertical strip, one third wide. The heights are one third, one, and one. So the area is one ninth plus one third plus one third: seven ninths, about zero point seven eight. The empty corner above the first strip is payment B's mistake. And A U C is this area, not the accuracy at any one threshold.", + "at": 316, + "duration": 56 + }, + { + "id": "pairs", + "narrator": "guide", + "text": "There's a second way to get the same answer. Pair every fraud with every legitimate payment, and give a point whenever the fraud scores higher. Three frauds times three legitimate payments makes nine pairs. A beats all three. C and D each beat two, and both lose to B. Seven wins out of nine: exactly seven ninths again.", + "at": 372, + "duration": 49 + }, + { + "id": "bridge", + "narrator": "guide", + "text": "Why do they agree? Take legitimate payment B, scored zero point eight. Only one of the three frauds outranks it, so its strip is one third tall. Legitimate payments E and F are outranked by all three frauds, so their strips are full height. Averaging the strip heights counts exactly the winning pairs. That's the bridge from area to probability.", + "at": 421, + "duration": 50 + }, + { + "id": "rigor", + "narrator": "guide", + "text": "Here is the exact statement. Pick one random fraud and one random legitimate payment. R O C area is the probability that the fraud scores higher, with ties counting half. On a finite sample, that is just the fraction of winning pairs, like our seven out of nine. And because the horizontal axis is false positive rate, that is what the area integrates over, not the threshold.", + "at": 471, + "duration": 55 + }, + { + "id": "ties", + "narrator": "guide", + "text": "Ties need a rule. Take a separate tiny example: one fraud and one legitimate payment, both scored zero point five. Any threshold flags both or neither, so the R O C curve jumps straight from zero, zero to one, one. That diagonal has area one half, the same as giving the tied pair half credit. So a model that gives everything the same score has an A U C of one half.", + "at": 526, + "duration": 58 + }, + { + "id": "baselines", + "narrator": "guide", + "text": "A perfect ranking climbs to the top before moving right, and fills the whole square. Random ordering has an expected area of one half. A completely reversed ranking scores zero. Below one half can signal a flipped score direction, but don't flip a model just because of your test set. And an A U C of zero point eight does not mean eighty percent of predictions are correct.", + "at": 584, + "duration": 56 + }, + { + "id": "calibration", + "narrator": "guide", + "text": "Now square every score. The numbers change, but the order doesn't, so the R O C curve and its area stay at seven ninths. A U C can't tell you which set of numbers makes better probabilities. A fixed cutoff has to move too: zero point six five becomes its square, about zero point four two. Check calibration separately, and pick the threshold from the costs of mistakes.", + "at": 640, + "duration": 56 + }, + { + "id": "imbalance", + "narrator": "guide", + "text": "Suppose a threshold catches eighty percent of frauds and flags ten percent of legitimate payments. With a hundred of each, that's eighty true alarms and ten false ones: about eighty nine percent precision. Now take ten frauds among nine hundred ninety legitimate payments. The same rates give eight true alarms and ninety nine false ones: only seven and a half percent precision. The R O C point never moved: eighty percent, ten percent. Precision did, because prevalence, the Greek letter pi, changes what the flagged pile is made of.", + "at": 696, + "duration": 70 + }, + { + "id": "pr", + "narrator": "guide", + "text": "Back to our six payments. Recall is just another name for the true positive rate. Precision asks a different question: of the payments we flagged, how many are really fraud? A precision recall curve plots recall across and precision up, as we sweep the same threshold. Each fraud we find raises recall; each legitimate payment we flag drags precision down. Before anything is flagged, precision is undefined, so plots usually start the curve at one.", + "at": 766, + "duration": 60 + }, + { + "id": "ap", + "narrator": "guide", + "text": "A common summary is average precision, or A P. Each new fraud raises recall by one third. At those moments, precision is one, then two thirds, then three quarters. Weight each by its recall step and add them up. Twenty-nine over thirty-six, about zero point eight one. Same predictions as our R O C area of seven ninths, but a different question, and a different number.", + "at": 826, + "duration": 54 + }, + { + "id": "conventions", + "narrator": "guide", + "text": "Average precision sums recall steps times precision: zero point eight oh six. Join the raw precision recall points with straight lines and take trapezoids, and you get zero point seven six four. Those straight lines aren't even a valid interpolation in precision recall space. Other benchmarks use an interpolated envelope: at each recall, the best precision at that recall or higher. Here it lifts the middle step from two thirds to three quarters. And on large samples, a random ranker's baseline precision is the prevalence: one half here, only because half our payments are fraud. Always name the convention.", + "at": 880, + "duration": 76 + }, + { + "id": "partial", + "narrator": "guide", + "text": "Say you can tolerate almost no false alarms. Partial A U C keeps only false positive rates up to zero point one. Here the raw area is one thirtieth, out of a possible one tenth. Scikit-learn's max F P R setting rescales that so one half is chance and one is perfect: about zero point six five. Always state the range, and whether it was rescaled.", + "at": 956, + "duration": 54 + }, + { + "id": "multiclass", + "narrator": "guide", + "text": "Now suppose a second model sorts confirmed frauds into three types. One versus rest turns each class into its own binary problem, with its own R O C area; one versus one compares pairs of classes instead. The macro average weights the classes equally: about zero point seven eight. Weighting by class size gives zero point nine one, because the biggest class is the easiest. Same model, two honest numbers with different priorities. And when an example can carry several labels, a micro average pools every example label pair into one curve; that is not an average of areas.", + "at": 1010, + "duration": 76 + }, + { + "id": "letters", + "narrator": "guide", + "text": "Outside classification, the same letters show up on different curves. In search, average precision rewards relevant results near the top, and mean average precision averages it over queries. With three relevant results, at ranks one, three, and four, it's our twenty-nine over thirty-six again. Object detection first matches each predicted box to a true one by their overlap, called I O U. These two boxes overlap by two thirds: a match at one half, a miss at three quarters. Only then does it build precision recall curves, so Coco's A P is not an R O C area. Survival models compute an A U C at each time horizon t. A machine that failed by t is a case, one still running at t is a control, and one last seen before t is unknown, not a negative. And area under a learning curve measures performance across a training budget, not ranking at all.", + "at": 1086, + "duration": 113 + }, + { + "id": "practice", + "narrator": "guide", + "text": "In practice, feed the metric scores, not hard labels, and make sure larger means more likely positive. Evaluate on held-out data, and report uncertainty that matches how you sampled. Look at the curve and at important subgroups, not just one number. Then choose a threshold from your costs and your prevalence. No kind of A U C chooses that for you.", + "at": 1199, + "duration": 51 + }, + { + "id": "recap", + "narrator": "guide", + "text": "Same six payments, two curves, and two different numbers: seven ninths, and twenty-nine over thirty-six. Neither is wrong; they answer different questions. R O C area asks whether frauds outrank legitimate payments. Precision recall asks how clean the flagged pile stays as recall grows. So when you hear A U C, ask: which curve, which positive class, which population, which range, and which average?", + "at": 1250, + "duration": 53 + } + ] +} diff --git a/videos/auc-ml/src/roc.tsx b/videos/auc-ml/src/roc.tsx new file mode 100644 index 0000000..4138796 --- /dev/null +++ b/videos/auc-ml/src/roc.tsx @@ -0,0 +1,916 @@ +import { ITEMS, NEG, POS, ROC, STATES } from "./data"; +import { + Arrow, + appear, + C, + Caption, + Card, + Chip, + clamp, + Eq, + ease, + In, + Plot, + Scores, + SVG, + Text, +} from "./visuals"; + +export type LineAt = (k: number, offsetSeconds?: number, fraction?: number) => number; +type SceneProps = { id: string; f: number; duration: number; L: LineAt }; +const slot = (i: number) => 76 + i * 186; // chip x for rank i + +/** Fraud × legitimate pair grid. `shown(i, j)` is 0..1 mark visibility, `hi(j)` highlights a column. */ +export function Pairs({ + x = 770, + y = 92, + cell = 84, + cellsIn = () => 1, + shown = () => 1, + hi = () => 0, + headers = 1, +}: { + x?: number; + y?: number; + cell?: number; + cellsIn?: (i: number, j: number) => number; + shown?: (i: number, j: number) => number; + hi?: (j: number) => number; + headers?: number; +}) { + const g = cell + 8; + return ( + + + + legitimate payment (−) + + {NEG.map((n, j) => ( + + {n.id} {n.score.toFixed(1)} + + ))} + {POS.map((p, i) => ( + + {p.id} {p.score.toFixed(1)} + + ))} + + {POS.map((p, i) => + NEG.map((n, j) => { + const win = p.score > n.score, + s = shown(i, j), + h = hi(j), + inP = cellsIn(i, j); + return ( + + 0 ? C.gold : C.line} + strokeWidth={1 + h * 2.5} + /> + + + {win ? "✓" : "✗"} + + + ); + }), + )} + + ); +} + +export function RocScenes({ id, f, L }: SceneProps) { + const at = (k: number, o = 0, fr = 0) => L(k, o, fr); + const A = (k: number, o = 0, dur = 14) => appear(f, at(k, o), dur); + + if (id === "intro") { + // Unsorted arrival → sorted queue → truth revealed → the one-number promise. + const order = [3, 5, 0, 4, 1, 2]; // display slot of rank i before sorting + const sort = ease((f - at(1, 1.2)) / 40), + drop = 150 * (1 - ease((f - at(3, 0)) / 26)); // queue sits centred until the AUC row arrives + return ( + <> + + + + + REVIEW FIRST + + + REVIEW LAST + + + + {ITEMS.map((item, i) => { + const p = appear(f, 8 + order[i] * 5, 16); + const xx = slot(order[i]) + (slot(i) - slot(order[i])) * sort, + // Chips moving right arc up, chips moving left arc down, so crossing scores stay legible. + lift = Math.sin(Math.PI * sort) * Math.sign(slot(order[i]) - slot(i)) * 64; + return ( + + + + ); + })} + + + B is legitimate, yet it ranks above two frauds. + + + + + + AUC + + + area under the curve + + + one number for the quality of the whole ordering + + + + + You'll compute it by hand from these six payments. + + ); + } + + if (id === "map") { + const cards = [ + { t: "Scores", s: "A .90 · B .80 · …", c: C.ink }, + { t: "Threshold", s: "flag if score ≥ t", c: C.gold }, + { t: "Curve", s: "one point per threshold", c: C.pos }, + { t: "Area", s: "one number for the curve", c: C.pos }, + ]; + return ( + + {cards.map((v, i) => ( + + + + + + + {i < 3 && } + + ))} + + + Same six payments, two kinds of curve + + + ROC curve + + + precision–recall curve + + + + ); + } + + if (id === "scores") { + const focus = A(3, 0.6, 18); + return ( + + {ITEMS.map((item, i) => ( + + + + ))} + + + + + higher score = more suspicious + + + order matters; the numbers need not be probabilities + + + + + + + + B outranks C and D + + + + + + → this mistake becomes missing area + + + + ); + } + + if (id === "threshold") { + const cutX = slot(3) - 13; + const cells = [ + { k: 2, label: "TP", who: "A, C", n: 2, col: 0, row: 0, c: C.pos }, + { k: 3, label: "FP", who: "B", n: 1, col: 0, row: 1, c: C.neg }, + { k: 4, label: "FN", who: "D", n: 1, col: 1, row: 0, c: C.pos }, + { k: 5, label: "TN", who: "E, F", n: 2, col: 1, row: 1, c: C.neg }, + ]; + const gx = 520, + gy = 240, + cw = 260, + ch = 82; + return ( + <> + + (i < 3 ? appear(f, at(1, 0.3) + i * 6, 10) : 0)} /> + + + + threshold 0.65 + + + + + flagged + + + not flagged + + + actual fraud + + + + actually legit − + + + {cells.map((v) => { + const p = A(v.k, 0.4), + x = gx + v.col * (cw + 10), + y = gy + v.row * (ch + 10); + return ( + + + + + + {v.label} + + + {v.who} + + + {v.n} + + + + ); + })} + + One threshold → one confusion matrix. Not yet a curve. + + ); + } + + if (id === "rates") { + const dots = (list: typeof ITEMS, hit: (id: string) => boolean, x: number, y: number, p: number) => + list.map((it, i) => ( + + + + {it.id} + + + )); + const pt = A(5, 0.3, 18); + return ( + <> + + + + Of the actual frauds, how many did we catch? + + + {dots(POS, (i) => i !== "D", 108, 80, A(0, 1.4))} + + + Of the legitimate payments, how many did we flag? + + + {dots(NEG, (i) => i === "B", 108, 250, A(2, 1.6))} + + + + + + (1/3, 2/3) + + + + + + + ROC = receiver operating characteristic: FPR across, TPR up. + + + ); + } + + if (id === "sweep") { + // The walk spans the two sentences that explain it, finishing on "At the bottom". + const k = clamp((f - at(2, 0.2)) / (at(3, 0.6) - at(2, 0.2))) * 6.3, + n = Math.min(6, Math.floor(k + 1e-6)), + step = n === 0 ? 0 : n - 1 + ease(clamp((k - n) / 0.3)), + s = STATES[n], + rowY = (i: number) => 52 + i * 52, + lineY = rowY(0) - 4 + step * 52; // rests in the gap below the last flagged row + return ( + <> + + + + + ROC curve + + + {ITEMS.map((item, i) => ( + + + + {item.id} {item.positive ? "+" : "−"} {item.score.toFixed(2)} + + + {i < n ? (item.positive ? "↑ up 1/3" : "→ right 1/3") : "not flagged"} + + + ))} + + + + threshold + + + + {`TPR ${s.tp}/3 · FPR ${s.fp}/3`} + + + {n === 0 + ? "above 0.90" + : n === 6 + ? "below 0.10" + : `between ${ITEMS[n - 1].score.toFixed(2)} and ${ITEMS[n].score.toFixed(2)}`} + + + Fraud (+): up 1/3. Legitimate (−): right 1/3. Both rates finish at 1. + + ); + } + + if (id === "area" || id === "bridge") { + const bridge = id === "bridge", + W = bridge ? 400 : 460, + H = bridge ? 230 : 300, + X = 100, + Y = 50; + const heights = [1 / 3, 1, 1]; + const fillOf = (j: number) => (bridge ? 1 : appear(f, at(2, 0.2) + j * 30, 18)); + const hiOf = (j: number) => (bridge ? (j === 0 ? A(1, 0.3) * (1 - A(3, 0)) : A(3, 0.4)) : 0); + return ( + <> + + + {heights.map((h, j) => ( + + + + + + ))} + {!bridge && ( + + + + + B's mistake + + + 2/9 missing + + + )} + + {bridge ? ( + <> + (j === 0 ? A(1, 0.3) * (1 - A(3, 0)) : A(3, 0.4))} /> + + + 1/3 tall + + + column B: 1 win of 3 + + + + + columns E, F: 3 wins of 3 + + + + ) : ( + <> + {["1/3 × 1/3 = 1/9", "1/3 × 1 = 1/3", "1/3 × 1 = 1/3"].map((t, j) => ( + + + {t} + + + {["strip B", "strip E", "strip F"][j]} + + + ))} + + + + = 7/9 ≈ 0.778 + + + + + + width 1/3 + + + + )} + + {bridge && ( + + )} + + {bridge + ? "Strip height = fraction of frauds that outrank that legitimate payment." + : "AUC is this area, not the accuracy at any one threshold. Both axes are fractions, so it is unitless."} + + + ); + } + + if (id === "pairs") { + const wins = POS.reduce( + (a, p, i) => a + NEG.reduce((b, n, j) => b + (p.score > n.score ? (markP(i, j) > 0.5 ? 1 : 0) : 0), 0), + 0, + ); + function markP(i: number, j: number) { + return i === 0 ? appear(f, at(3, 0.2) + j * 6, 10) : appear(f, at(4, 0.2) + (i - 1) * 14 + j * 5, 10); + } + return ( + <> + + + + 3 frauds + + + × 3 legitimate + + + = 9 pairs + + + Math.max(0.35 * A(0, 0.6), appear(f, at(2, 0.4) + (i * 3 + j) * 3, 10))} + shown={markP} + hi={(j) => (j === 0 ? A(4, 1.8) : 0)} + /> + + + {wins} {wins === 1 ? "win" : "wins"} + + + + + ↑ both losses: B + + + + + 7 / 9 pairs + + + = ROC area 7/9 + + + + + ); + } + + if (id === "rigor") { + const focus = (k: number) => A(k, 0) * (1 - 0.6 * A(k + 1, 0)); + return ( + <> + + + + + > ? + + + + random fraud + + + random legitimate + + + + S_-)+\tfrac12\,P(S_+=S_-)`} /> + + + + ); + } + + if (id === "ties") { + const draw = ease((f - at(2, 1.2)) / 40); + return ( + <> + + + + + + + + 1/2 + + + credit for the tied pair + + + + + + area 1/2 + + + + + All scores tied → AUC = 1/2 (when both classes are present). + + + ); + } + + if (id === "baselines") { + const panels = [ + { + name: "Perfect ranking", + a: "AUC = 1", + points: [ + [0, 0], + [0, 1], + [1, 1], + ], + c: C.pos, + }, + { + name: "Random order", + a: "expected AUC = 0.5", + points: [ + [0, 0], + [1, 1], + ], + c: C.gold, + }, + { + name: "Reversed ranking", + a: "AUC = 0", + points: [ + [0, 0], + [1, 0], + [1, 1], + ], + c: C.neg, + }, + ]; + return ( + <> + + {panels.map((v, i) => { + const p = A(i, 0.2), + d = ease((f - at(i, 0.4)) / 40) * (v.points.length - 1); + return ( + + + {v.name} + + + + {v.a} + + + ); + })} + + + Below 0.5 may mean a flipped score; don't flip a model because of the test set. + + + AUC 0.8 ≠ “80% of predictions are correct.” + + + ); + } + + if (id === "calibration") { + const cutX = slot(3) - 13, + cut = A(3, 0.2); + return ( + <> + + + + + {ITEMS.map((_, i) => ( + + ))} + + + s → s² + + + + + + Same order → same ROC curve → AUC still 7/9 + + + + + + + + cutoff 0.65 → 0.65² = 0.4225 + + + + + Check calibration separately; choose the threshold from the costs of mistakes. + + + ); + } + return null; +} diff --git a/videos/auc-ml/src/score.ts b/videos/auc-ml/src/score.ts new file mode 100644 index 0000000..1fe3a60 --- /dev/null +++ b/videos/auc-ml/src/score.ts @@ -0,0 +1,199 @@ +import { defineScore, type Note, note } from "@archastro/clapper-music"; +import beats from "./beats.json"; +import timing from "./timing.json"; + +// A quiet bed for a narrated lesson. 60 BPM, so one beat is one second and every +// position below reads directly in film seconds. Music rises in the title card, +// the breaths between chapters and the end card, and sits well under speech. +type Timing = Record; +type Beat = { id: string; chapter?: string }; +const T = timing as Timing; +const B = beats as Beat[]; +const scenes: { id: string; part: number; start: number; seconds: number; speech?: [number, number] }[] = []; +{ + let t = 0, + part = 0; + for (const b of B) { + const s = T[b.id] ?? { seconds: 8, lines: [] }; + if (b.chapter) part = Number(b.chapter.slice(0, 2)) - 1; + const first = s.lines[0], + last = s.lines.at(-1); + scenes.push({ + id: b.id, + part, + start: t, + seconds: s.seconds, + speech: first && last ? [t + first.at, t + last.at + last.dur] : undefined, + }); + t += s.seconds; + } +} +const TOTAL = scenes.reduce((a, s) => a + s.seconds, 0); +const sceneAt = (t: number) => [...scenes].reverse().find((s) => s.start <= t) ?? scenes[0]; +const startOf = (id: string) => scenes.find((s) => s.id === id)?.start ?? 0; + +const NAMES = ["C", "C#", "D", "D#", "E", "F", "F#", "G", "G#", "A", "A#", "B"]; +const midi = (p: string) => { + const m = /^([A-G]#?)(-?\d)$/.exec(p)!; + return NAMES.indexOf(m[1]) + 12 * (Number(m[2]) + 1); +}; + +// Dmaj9 → Bm11 → Gmaj7(#11) → Asus: slow and open, no strong cadence until the end card. +const HARMONY = [ + { bass: "D2", pad: ["A3", "D4", "F#4"], top: ["E5", "F#5", "A5"] }, + { bass: "B2", pad: ["F#3", "D4", "E4"], top: ["D5", "F#5", "A5"] }, + { bass: "G2", pad: ["B3", "D4", "F#4"], top: ["C#5", "F#5", "B5"] }, + { bass: "A2", pad: ["E3", "A3", "D4"], top: ["B4", "E5", "A5"] }, +]; +const BAR = 8; // beats (= seconds) per chord +// Part 4 (precision–recall) lifts a whole step: a new curve, a new colour. +const shift = (t: number) => (sceneAt(t).part === 3 ? 2 : 0); +const END = startOf("end"); +const LOOP_END = END - 4; // then an explicit A → D cadence on the end card + +const piano: Note[] = [], + strings: Note[] = [], + bass: Note[] = [], + harp: Note[] = []; +// The piano figure changes with each part of the lesson instead of every bar: [offset, top index, length]. +const FIGURES = [ + [ + [0, 0, 3], + [1.5, 1, 2.5], + [3, 2, 4], + ], // intuition: rising broken chord + [ + [0, 2, 3], + [2, 1, 3], + [4, 0, 3.5], + ], // build ROC: the same notes, falling + [ + [0, 0, 3], + [3, 2, 4], + ], // ranking math: sparser, leaves room for formulas + [ + [0, 1, 2], + [1, 2, 2], + [2.5, 0, 3], + ], // precision–recall: a new shape in the new key + [[0, 2, 4]], // other contexts: one bell-like note per chord + [ + [0, 0, 3], + [1.5, 1, 2.5], + [3, 2, 4], + ], // use it: the opening figure returns +]; +for (let i = 0; i * BAR < LOOP_END; i++) { + const h = HARMONY[i % 4], + at = i * BAR, + len = Math.min(BAR, LOOP_END - at), + s = shift(at), + scene = sceneAt(at), + // Softer under the opening hook and the one-note "bells" of the survey part. + v = 36 + (i % 2) * 4 - (scene.part === 0 ? 5 : scene.part === 4 ? 8 : 0), + p = (name: string) => midi(name) + s; + for (const [dt, idx, d] of FIGURES[scene.part]) + if (dt < len) piano.push(note(p(h.top[idx]), at + dt, Math.min(d, len - dt), v - idx * 2)); + if (i % 2 && scene.part !== 2 && len > 5) + piano.push(note(p(h.pad[2]), at + 5, Math.min(3, len - 5), v - 8)); + // Strings rest under the dense survey of other curves, then return. + // Rest only for bars wholly inside it, so the pad is back as soon as the next scene starts. + if (scene.id !== "letters" || sceneAt(at + len - 0.01).id !== "letters") + strings.push(...h.pad.map((n) => note(p(n), at, len, 44))); + bass.push(note(p(h.bass), at, len, 40)); +} +// Cadence: A (with its third) for four seconds, resolving to D as the end card lands. +strings.push(...["E3", "A3", "C#4"].map((n) => note(n, LOOP_END, 4, 46))); +bass.push(note("A2", LOOP_END, 4, 42)); +piano.push(note("C#5", LOOP_END + 0.5, 3, 34), note("E5", LOOP_END + 2, 2, 32)); +strings.push(...["D3", "A3", "F#4"].map((n) => note(n, END, 9, 48))); +bass.push(note("D2", END, 9, 44)); +piano.push(...["D4", "F#4", "A4", "D5"].map((n, k) => note(n, END + 0.1 + k * 0.05, 6, 44 - k * 2))); +harp.push(...["D4", "A4", "D5", "F#5", "A5"].map((n, k) => note(n, END + 0.3 + k * 0.22, 6, 54 - k * 3))); + +// Title card: a rising figure that lands on the wordmark. +const title = startOf("title"); +["D5", "F#5", "A5", "C#6", "E6"].forEach((n, k) => + harp.push(note(n, title + 0.4 + k * 0.32, 3.5, 58 - k * 3)), +); +// Each new part gets one soft harp glint on its first cut, taken from the chord sounding then. +B.forEach((b, i) => { + if (i === 0 || !b.chapter || b.chapter.slice(0, 2) === B[i - 1].chapter?.slice(0, 2)) return; + const t = startOf(b.id), + h = HARMONY[Math.floor(t / BAR) % 4]; + harp.push(note(midi(h.top[2]) + shift(t), t, 4, 46)); +}); + +// Ducking: full on the title and end cards, a breath between chapters, low under speech. +const DUCK = 0.3, + GAP = 0.6, + CARD = 1.4, + RAMP = 0.7; +const auto: { at: number; value: number }[] = [ + { at: 0, value: 0 }, + { at: 0.6, value: GAP }, +]; +for (const s of scenes) { + if (!s.speech) { + auto.push( + { at: s.start + 0.3, value: CARD }, + { at: s.start + s.seconds - (s.id === "end" ? 4 : 1.2), value: CARD }, + ); + continue; + } + const [a, b] = s.speech; + auto.push({ at: Math.max(a - RAMP, auto.at(-1)!.at + 0.05), value: GAP }); + auto.push({ at: a, value: DUCK }); + auto.push({ at: b + 0.2, value: DUCK }); + auto.push({ at: b + 0.2 + RAMP * 1.5, value: GAP }); +} +auto.push({ at: TOTAL - 0.05, value: 0 }); +// Deduplicate positions that collide when one scene's speech follows closely after another's. +const gain = auto.sort((x, y) => x.at - y.at).filter((p, i, a) => i === 0 || p.at - a[i - 1].at > 0.04); + +export default defineScore({ + title: "AUC — study bed", + tempo: 60, + meter: [4, 4], + seed: 7, + length: TOTAL + 6, + tail: 4, + tracks: [ + { + id: "piano", + instrument: "vsupright1", + gain: 0.75, + reverb: 0.45, + pan: -0.12, + clips: [{ notes: piano }], + humanize: { timing: 0.012, velocity: 4 }, + gainAutomation: gain, + }, + { + id: "strings", + instrument: "viola-ens-sus-vib-quiet", + gain: 0.17, + reverb: 0.5, + pan: 0.15, + clips: [{ notes: strings }], + gainAutomation: gain, + }, + { + id: "bass", + instrument: "cello-ens-sus-vib-quiet", + gain: 0.12, + reverb: 0.35, + clips: [{ notes: bass }], + gainAutomation: gain, + }, + { + id: "harp", + instrument: "harp", + gain: 0.6, + reverb: 0.55, + pan: 0.1, + clips: [{ notes: harp }], + gainAutomation: gain, + }, + ], +}); diff --git a/videos/auc-ml/src/theme.css b/videos/auc-ml/src/theme.css new file mode 100644 index 0000000..171dcd9 --- /dev/null +++ b/videos/auc-ml/src/theme.css @@ -0,0 +1,133 @@ +@import "./fonts/fonts.css"; +* { + box-sizing: border-box; +} +.film { + position: absolute; + inset: 0; + background: #f4f1e8; + color: #193239; + overflow: hidden; + font-family: "Schibsted Grotesk", sans-serif; + -webkit-font-smoothing: antialiased; +} +/* Faint paper grid: a quiet reference surface for every diagram. */ +.film::before { + content: ""; + position: absolute; + inset: 0; + background-image: + linear-gradient(#193239 1px, transparent 1px), linear-gradient(90deg, #193239 1px, transparent 1px); + background-size: 48px 48px; + opacity: 0.035; +} +.stage { + width: 1280px; + height: 720px; + position: absolute; + transform: scale(1.5); + transform-origin: 0 0; +} +.chapter { + position: absolute; + left: 76px; + top: 30px; + color: #0b6d60; + font: + 500 15px "Fragment Mono", + monospace; + letter-spacing: 1.4px; + text-transform: uppercase; +} +.title { + position: absolute; + left: 76px; + top: 54px; + width: 1130px; + margin: 0; + line-height: 1.08; + font-size: 42px; + font-weight: 650; + letter-spacing: -1.1px; +} +.subtitle { + position: absolute; + left: 77px; + top: 112px; + max-width: 1120px; + font-size: 22px; + line-height: 1.25; + color: #3d5156; + margin: 0; +} +.canvas { + position: absolute; + top: 168px; + left: 0; + width: 1280px; + height: 490px; +} +.caption { + position: absolute; + left: 76px; + right: 76px; + top: 432px; + padding: 7px 16px; + background: #e6ece2; + border-left: 4px solid #0b6d60; + font-size: 24px; + line-height: 1.22; +} +.equation .katex-display { + margin: 0; +} +.equation .katex { + font-size: 1em; +} +/* Film-wide chapter rail: where we are in the lesson. */ +.rail { + position: absolute; + left: 76px; + right: 76px; + bottom: 10px; + height: 30px; + display: flex; + gap: 8px; +} +.rail .seg { + position: relative; + flex: 1; + font: + 500 18px "Fragment Mono", + monospace; + letter-spacing: 0.5px; + text-transform: uppercase; + color: #8a9894; + padding-top: 8px; +} +.rail .seg::before { + content: ""; + position: absolute; + top: 0; + left: 0; + right: 0; + height: 2px; + background: #d6ddd6; +} +.rail .seg .fill { + position: absolute; + top: 0; + left: 0; + height: 2px; + background: #0b6d60; +} +.rail .seg.active { + color: #0b6d60; +} +.serif { + font-family: "Instrument Serif", serif; + font-weight: 400; +} +.mono { + font-family: "Fragment Mono", monospace; +} diff --git a/videos/auc-ml/src/timing.json b/videos/auc-ml/src/timing.json new file mode 100644 index 0000000..88d655d --- /dev/null +++ b/videos/auc-ml/src/timing.json @@ -0,0 +1,720 @@ +{ + "intro": { + "seconds": 25, + "take": { + "at": 0.6, + "dur": 22.525 + }, + "lines": [ + { + "at": 0.9, + "dur": 2.74 + }, + { + "at": 3.64, + "dur": 4.27 + }, + { + "at": 7.91, + "dur": 4.17 + }, + { + "at": 12.08, + "dur": 4.68 + }, + { + "at": 16.76, + "dur": 5.91 + } + ] + }, + "title": { + "seconds": 7, + "lines": [] + }, + "map": { + "seconds": 26, + "take": { + "at": 0.6, + "dur": 23.35 + }, + "lines": [ + { + "at": 0.88, + "dur": 2.1 + }, + { + "at": 2.98, + "dur": 3.6 + }, + { + "at": 6.58, + "dur": 3.99 + }, + { + "at": 10.57, + "dur": 3.45 + }, + { + "at": 14.02, + "dur": 4.38 + }, + { + "at": 18.4, + "dur": 5.05 + } + ] + }, + "scores": { + "seconds": 27, + "take": { + "at": 0.6, + "dur": 24.425 + }, + "lines": [ + { + "at": 0.91, + "dur": 3.92 + }, + { + "at": 4.83, + "dur": 3.99 + }, + { + "at": 8.82, + "dur": 5.77 + }, + { + "at": 14.59, + "dur": 5.64 + }, + { + "at": 20.23, + "dur": 1.7 + }, + { + "at": 21.93, + "dur": 2.6 + } + ] + }, + "threshold": { + "seconds": 26.5, + "take": { + "at": 0.6, + "dur": 23.75 + }, + "lines": [ + { + "at": 0.89, + "dur": 2.91 + }, + { + "at": 3.8, + "dur": 2.1 + }, + { + "at": 5.9, + "dur": 3.1 + }, + { + "at": 9, + "dur": 4.13 + }, + { + "at": 13.13, + "dur": 3.3 + }, + { + "at": 16.43, + "dur": 3.94 + }, + { + "at": 20.37, + "dur": 3.5 + } + ] + }, + "rates": { + "seconds": 30, + "take": { + "at": 0.6, + "dur": 27.125 + }, + "lines": [ + { + "at": 0.89, + "dur": 5.75 + }, + { + "at": 6.64, + "dur": 1.53 + }, + { + "at": 8.17, + "dur": 6.7 + }, + { + "at": 14.87, + "dur": 1.66 + }, + { + "at": 16.53, + "dur": 8.64 + }, + { + "at": 25.17, + "dur": 2.1 + } + ] + }, + "sweep": { + "seconds": 24.5, + "take": { + "at": 0.6, + "dur": 21.65 + }, + "lines": [ + { + "at": 0.87, + "dur": 5.18 + }, + { + "at": 6.05, + "dur": 3.51 + }, + { + "at": 9.56, + "dur": 6.07 + }, + { + "at": 15.63, + "dur": 4 + }, + { + "at": 19.63, + "dur": 2.16 + } + ] + }, + "area": { + "seconds": 29.5, + "take": { + "at": 0.6, + "dur": 25.75 + }, + "lines": [ + { + "at": 0.91, + "dur": 2.45 + }, + { + "at": 3.36, + "dur": 4.33 + }, + { + "at": 7.69, + "dur": 2.79 + }, + { + "at": 10.48, + "dur": 7 + }, + { + "at": 17.48, + "dur": 4.66 + }, + { + "at": 22.14, + "dur": 3.74 + } + ] + }, + "pairs": { + "seconds": 25.5, + "take": { + "at": 0.6, + "dur": 21.825 + }, + "lines": [ + { + "at": 0.93, + "dur": 2.79 + }, + { + "at": 3.72, + "dur": 6 + }, + { + "at": 9.72, + "dur": 4.17 + }, + { + "at": 13.89, + "dur": 1.75 + }, + { + "at": 15.64, + "dur": 3.22 + }, + { + "at": 18.86, + "dur": 3.07 + } + ] + }, + "bridge": { + "seconds": 27, + "take": { + "at": 0.6, + "dur": 23.25 + }, + "lines": [ + { + "at": 0.91, + "dur": 1.67 + }, + { + "at": 2.58, + "dur": 3.79 + }, + { + "at": 6.37, + "dur": 4.77 + }, + { + "at": 11.14, + "dur": 6.08 + }, + { + "at": 17.22, + "dur": 3.88 + }, + { + "at": 21.1, + "dur": 2.25 + } + ] + }, + "rigor": { + "seconds": 29.5, + "take": { + "at": 0.6, + "dur": 25.75 + }, + "lines": [ + { + "at": 0.9, + "dur": 2.3 + }, + { + "at": 3.2, + "dur": 4.23 + }, + { + "at": 7.43, + "dur": 5.9 + }, + { + "at": 13.33, + "dur": 5.76 + }, + { + "at": 19.09, + "dur": 6.76 + } + ] + }, + "ties": { + "seconds": 30, + "take": { + "at": 0.6, + "dur": 26.375 + }, + "lines": [ + { + "at": 0.92, + "dur": 1.72 + }, + { + "at": 2.64, + "dur": 6.98 + }, + { + "at": 9.62, + "dur": 7.46 + }, + { + "at": 17.08, + "dur": 5.28 + }, + { + "at": 22.36, + "dur": 4.14 + } + ] + }, + "baselines": { + "seconds": 27, + "take": { + "at": 0.6, + "dur": 24.25 + }, + "lines": [ + { + "at": 0.88, + "dur": 5.11 + }, + { + "at": 5.99, + "dur": 3.72 + }, + { + "at": 9.71, + "dur": 3.09 + }, + { + "at": 12.8, + "dur": 6.56 + }, + { + "at": 19.36, + "dur": 5.01 + } + ] + }, + "calibration": { + "seconds": 28, + "take": { + "at": 0.6, + "dur": 25.25 + }, + "lines": [ + { + "at": 0.91, + "dur": 2.01 + }, + { + "at": 2.92, + "dur": 6.39 + }, + { + "at": 9.31, + "dur": 4.65 + }, + { + "at": 13.96, + "dur": 7.15 + }, + { + "at": 21.11, + "dur": 4.31 + } + ] + }, + "imbalance": { + "seconds": 39, + "take": { + "at": 0.6, + "dur": 34.925 + }, + "lines": [ + { + "at": 0.88, + "dur": 6.7 + }, + { + "at": 7.58, + "dur": 6.57 + }, + { + "at": 14.15, + "dur": 4.34 + }, + { + "at": 18.49, + "dur": 6.74 + }, + { + "at": 25.23, + "dur": 4.42 + }, + { + "at": 29.65, + "dur": 5.4 + } + ] + }, + "pr": { + "seconds": 33.5, + "take": { + "at": 0.6, + "dur": 30.775 + }, + "lines": [ + { + "at": 0.88, + "dur": 2.09 + }, + { + "at": 2.97, + "dur": 4.03 + }, + { + "at": 7, + "dur": 5.92 + }, + { + "at": 12.92, + "dur": 6.4 + }, + { + "at": 19.32, + "dur": 6.34 + }, + { + "at": 25.66, + "dur": 5.22 + } + ] + }, + "ap": { + "seconds": 29.5, + "take": { + "at": 0.6, + "dur": 25.6 + }, + "lines": [ + { + "at": 0.88, + "dur": 3.5 + }, + { + "at": 4.38, + "dur": 3.24 + }, + { + "at": 7.62, + "dur": 4.73 + }, + { + "at": 12.35, + "dur": 3.28 + }, + { + "at": 15.63, + "dur": 4.2 + }, + { + "at": 19.83, + "dur": 5.86 + } + ] + }, + "conventions": { + "seconds": 45.5, + "take": { + "at": 0.6, + "dur": 41.575 + }, + "lines": [ + { + "at": 0.92, + "dur": 5.57 + }, + { + "at": 6.49, + "dur": 7.86 + }, + { + "at": 14.35, + "dur": 5.66 + }, + { + "at": 20.01, + "dur": 7.19 + }, + { + "at": 27.2, + "dur": 3.98 + }, + { + "at": 31.18, + "dur": 9.11 + }, + { + "at": 40.29, + "dur": 1.41 + } + ] + }, + "partial": { + "seconds": 28.5, + "take": { + "at": 0.6, + "dur": 24.85 + }, + "lines": [ + { + "at": 0.9, + "dur": 3.12 + }, + { + "at": 4.02, + "dur": 5.32 + }, + { + "at": 9.34, + "dur": 4.37 + }, + { + "at": 13.71, + "dur": 8.55 + }, + { + "at": 22.26, + "dur": 2.72 + } + ] + }, + "multiclass": { + "seconds": 43.5, + "take": { + "at": 0.6, + "dur": 39.675 + }, + "lines": [ + { + "at": 0.91, + "dur": 4.73 + }, + { + "at": 5.64, + "dur": 10.21 + }, + { + "at": 15.85, + "dur": 5.09 + }, + { + "at": 20.94, + "dur": 6.02 + }, + { + "at": 26.96, + "dur": 3.9 + }, + { + "at": 30.86, + "dur": 8.93 + } + ] + }, + "letters": { + "seconds": 60.5, + "take": { + "at": 0.6, + "dur": 58.025 + }, + "lines": [ + { + "at": 0.93, + "dur": 4.55 + }, + { + "at": 5.48, + "dur": 7.93 + }, + { + "at": 13.41, + "dur": 6.54 + }, + { + "at": 19.95, + "dur": 6.37 + }, + { + "at": 26.32, + "dur": 6.33 + }, + { + "at": 32.65, + "dur": 6.47 + }, + { + "at": 39.12, + "dur": 4.51 + }, + { + "at": 43.63, + "dur": 9.14 + }, + { + "at": 52.77, + "dur": 5.32 + } + ] + }, + "practice": { + "seconds": 26, + "take": { + "at": 0.6, + "dur": 23.275 + }, + "lines": [ + { + "at": 0.93, + "dur": 6.97 + }, + { + "at": 7.9, + "dur": 5.59 + }, + { + "at": 13.49, + "dur": 4.11 + }, + { + "at": 17.6, + "dur": 3.77 + }, + { + "at": 21.37, + "dur": 2.01 + } + ] + }, + "recap": { + "seconds": 29.5, + "take": { + "at": 0.6, + "dur": 26.975 + }, + "lines": [ + { + "at": 0.9, + "dur": 6.86 + }, + { + "at": 7.76, + "dur": 3.27 + }, + { + "at": 11.03, + "dur": 4.6 + }, + { + "at": 15.63, + "dur": 4.93 + }, + { + "at": 20.56, + "dur": 6.53 + } + ] + }, + "end": { + "seconds": 10, + "lines": [] + } +} diff --git a/videos/auc-ml/src/visuals.tsx b/videos/auc-ml/src/visuals.tsx new file mode 100644 index 0000000..31a792f --- /dev/null +++ b/videos/auc-ml/src/visuals.tsx @@ -0,0 +1,459 @@ +import { Latex } from "@archastro/clapper-core/latex"; +import type { CSSProperties, ReactNode } from "react"; +import { ITEMS, ROC } from "./data"; +export const C = { + bg: "#f4f1e8", + paper: "#fffefa", + ink: "#193239", + muted: "#3d5156", + faint: "#8a9894", + line: "#d6ddd6", + pos: "#006657", + posSoft: "#d9ede6", + neg: "#a3402b", + negSoft: "#f5ded5", + gold: "#8a5a00", + goldSoft: "#f6e7c4", + blue: "#24558e", + blueSoft: "#e0e9f4", + purple: "#624776", +}; +export const clamp = (x: number) => Math.max(0, Math.min(1, x)); +export const ease = (x: number) => { + const p = clamp(x); + return p * p * (3 - 2 * p); +}; +export const easeOut = (x: number) => 1 - (1 - clamp(x)) ** 3; +/** Eased 0→1 from frame `at` over `dur` frames. */ +export const appear = (f: number, at: number, dur = 14) => easeOut((f - at) / dur); +/** Fractional tick labels without vulgar-fraction glyphs (not in the house fonts). */ +export const fracLabel = (v: number) => + v === 0 + ? "0" + : v === 1 + ? "1" + : Math.abs(v - 1 / 3) < 1e-9 + ? "1/3" + : Math.abs(v - 2 / 3) < 1e-9 + ? "2/3" + : String(v); + +/** Fade + rise wrapper for SVG groups. */ +export function In({ + f, + at, + dur = 14, + dy = 10, + children, +}: { + f: number; + at: number; + dur?: number; + dy?: number; + children: ReactNode; +}) { + const p = appear(f, at, dur); + return ( + + {children} + + ); +} +/** Same, for HTML overlays. */ +export function HIn({ + f, + at, + dur = 14, + style, + children, +}: { + f: number; + at: number; + dur?: number; + style?: CSSProperties; + children: ReactNode; +}) { + const p = appear(f, at, dur); + return
{children}
; +} +export function Text({ + x, + y, + children, + size = 24, + color = C.ink, + anchor = "start", + weight = 560, + mono = false, + opacity, +}: { + x: number; + y: number; + children: ReactNode; + size?: number; + color?: string; + anchor?: "start" | "middle" | "end"; + weight?: number; + mono?: boolean; + opacity?: number; +}) { + return ( + + {children} + + ); +} +export function Card({ + x, + y, + w = 300, + h = 115, + title, + sub, + color = C.pos, + fill = C.paper, +}: { + x: number; + y: number; + w?: number; + h?: number; + title: ReactNode; + sub?: ReactNode; + color?: string; + fill?: string; +}) { + return ( + + + + + {title} + + {sub && ( + +
+ {sub} +
+
+ )} +
+ ); +} +export function Arrow({ + x1, + y1, + x2, + y2, + color = C.muted, + p = 1, + width = 2.5, +}: { + x1: number; + y1: number; + x2: number; + y2: number; + color?: string; + p?: number; + width?: number; +}) { + if (p <= 0) return null; + const ex = x1 + (x2 - x1) * p, + ey = y1 + (y2 - y1) * p; + const a = Math.atan2(y2 - y1, x2 - x1), + s = 11; + return ( + + + + + ); +} +export function Eq({ + x = 90, + y = 350, + w = 1100, + tex, + size = 30, + p = 1, + color = C.ink, +}: { + x?: number; + y?: number; + w?: number; + tex: string; + size?: number; + p?: number; + color?: string; +}) { + if (p <= 0) return null; + return ( +
+ + {tex} + +
+ ); +} +export function SVG({ children }: { children: ReactNode }) { + return ( + + {children} + + ); +} +/** One payment chip. */ +export function Chip({ + item, + x, + y, + w = 150, + h = 104, + value, + truth = 1, + flagged = 0, + dim = 0, + label = true, +}: { + item: (typeof ITEMS)[number]; + x: number; + y: number; + w?: number; + h?: number; + value?: number; + truth?: number; + flagged?: number; + dim?: number; + label?: boolean; +}) { + const c = item.positive ? C.pos : C.neg, + soft = item.positive ? C.posSoft : C.negSoft; + return ( + + + + + {label && ( + <> + + {item.id} + + + {item.positive ? "+ fraud" : "− legit"} + + + )} + + {(value ?? item.score).toFixed(2)} + + + ); +} +/** The six payments in rank order. */ +export function Scores({ + x = 76, + y = 40, + gap = 186, + w = 160, + squared = false, + truth = 1, + flagged = () => 0, + dim = () => 0, +}: { + x?: number; + y?: number; + gap?: number; + w?: number; + squared?: boolean; + truth?: number; + flagged?: (i: number) => number; + dim?: (i: number) => number; +}) { + return ( + + {ITEMS.map((item, i) => ( + + ))} + + ); +} +/** Path through points, revealed up to a fractional vertex index. */ +export function partialPath( + pts: [number, number][], + upto: number, + px: (v: number) => number, + py: (v: number) => number, +) { + if (upto <= 0 || pts.length === 0) return { d: "", end: pts[0] ?? [0, 0] }; + const n = Math.min(Math.floor(upto), pts.length - 1), + frac = upto - n; + const shown = pts.slice(0, n + 1).map((p) => [...p] as [number, number]); + if (frac > 0 && n + 1 < pts.length) { + const a = pts[n], + b = pts[n + 1]; + shown.push([a[0] + (b[0] - a[0]) * frac, a[1] + (b[1] - a[1]) * frac]); + } + return { + d: shown.map(([a, b], i) => `${i ? "L" : "M"}${px(a)} ${py(b)}`).join(" "), + end: shown.at(-1)!, + }; +} +export function Plot({ + x = 96, + y = 46, + w = 440, + h = 290, + points = ROC, + upto = points.length - 1, + color = C.pos, + fill = 0, + diagonal = 0, + xLabel = "False-positive rate", + yLabel = "True-positive rate", + dot = true, + ticks = [0, 1 / 3, 2 / 3, 1], + axes = 1, + children, +}: { + x?: number; + y?: number; + w?: number; + h?: number; + points?: [number, number][]; + upto?: number; + color?: string; + fill?: number; + diagonal?: number; + xLabel?: string; + yLabel?: string; + dot?: boolean; + ticks?: number[]; + axes?: number; + children?: ReactNode; +}) { + const px = (v: number) => x + v * w, + py = (v: number) => y + (1 - v) * h, + { d, end } = partialPath(points, upto, px, py); + return ( + + + + {ticks.map((v) => ( + + + + {fracLabel(v)} + + + {fracLabel(v)} + + + ))} + {yLabel && ( + + {yLabel} ↑ + + )} + {xLabel && ( + + {xLabel} → + + )} + + {diagonal > 0 && ( + + )} + {fill > 0 && d && ( + + )} + {children} + {d && ( + + )} + {dot && d && ( + + )} + + + ); +} +export function Caption({ + children, + color = C.pos, + p = 1, +}: { + children: ReactNode; + color?: string; + p?: number; +}) { + if (p <= 0) return null; + return ( +
+ {children} +
+ ); +} diff --git a/videos/auc-ml/transcript.md b/videos/auc-ml/transcript.md new file mode 100644 index 0000000..4cdac69 --- /dev/null +++ b/videos/auc-ml/transcript.md @@ -0,0 +1,87 @@ +## 00:00.0 — Who should be reviewed first? + +A fraud model scores every payment. You can only review a few today, so the highest scores go first. But how do you know that ordering is any good, before you pick a cutoff? One number answers that: the area under a curve, or A U C. You'll compute it by hand from these six payments, then see why the same three letters can mean very different things. + +## 00:32.0 — Scores → thresholds → curve → area + +Here is the whole journey. On held-out examples, the model produces scores. A threshold turns each score into a yes or no decision. Move the threshold, and each setting gives one point. Plot those points as a curve, then summarize the curve with its area. We'll use the same six payments for two kinds of curve, so you can see exactly what changes. + +## 00:58.0 — A score gives an ordering + +Here are the six payments, sorted from highest score to lowest. Plus means actual fraud; minus means legitimate. The scores only need to put payments in order; they need not be calibrated probabilities. Notice one mistake: payment B is legitimate, yet it ranks above two frauds, C and D. Keep B in mind. Its mistake will become a missing piece of area. + +## 01:25.0 — One threshold gives one decision + +Set the threshold at zero point six five. We flag A, B, and C. A and C are frauds we caught: true positives. B is a legitimate payment we flagged: a false positive. D is a fraud we missed: a false negative. And E and F are correctly left alone: true negatives. One threshold gives one confusion matrix, not yet a curve. + +## 01:51.5 — Two rates, two different denominators + +The true positive rate asks: of all the actual frauds, what fraction did we catch? Two out of three. The false positive rate asks: of all the legitimate payments, what fraction did we wrongly flag? One out of three. An R O C curve, short for receiver operating characteristic, plots false positive rate across and true positive rate up. Our threshold becomes this single point. + +## 02:21.5 — Lower the threshold. Trace the tradeoff. + +Start above the highest score: nothing is flagged, so both rates are zero. Now lower the threshold past one payment at a time. Each fraud steps the curve up by a third; each legitimate payment steps it right by a third. At the bottom, everything is flagged, and we reach one, one. That staircase is the R O C curve. + +## 02:46.0 — Width × height, added across the curve + +Now shade the area under the curve. Each legitimate payment adds a vertical strip, one third wide. The heights are one third, one, and one. So the area is one ninth plus one third plus one third: seven ninths, about zero point seven eight. The empty corner above the first strip is payment B's mistake. And A U C is this area, not the accuracy at any one threshold. + +## 03:15.5 — Area is also a ranking game + +There's a second way to get the same answer. Pair every fraud with every legitimate payment, and give a point whenever the fraud scores higher. Three frauds times three legitimate payments makes nine pairs. A beats all three. C and D each beat two, and both lose to B. Seven wins out of nine: exactly seven ninths again. + +## 03:41.0 — Each strip is one column of comparisons + +Why do they agree? Take legitimate payment B, scored zero point eight. Only one of the three frauds outranks it, so its strip is one third tall. Legitimate payments E and F are outranked by all three frauds, so their strips are full height. Averaging the strip heights counts exactly the winning pairs. That's the bridge from area to probability. + +## 04:08.0 — The probability behind ROC-AUC + +Here is the exact statement. Pick one random fraud and one random legitimate payment. R O C area is the probability that the fraud scores higher, with ties counting half. On a finite sample, that is just the fraction of winning pairs, like our seven out of nine. And because the horizontal axis is false positive rate, that is what the area integrates over, not the threshold. + +## 04:37.5 — Equal scores mean no ranking preference + +Ties need a rule. Take a separate tiny example: one fraud and one legitimate payment, both scored zero point five. Any threshold flags both or neither, so the R O C curve jumps straight from zero, zero to one, one. That diagonal has area one half, the same as giving the tied pair half credit. So a model that gives everything the same score has an A U C of one half. + +## 05:07.5 — 1 is perfect. 0.5 is the chance reference. + +A perfect ranking climbs to the top before moving right, and fills the whole square. Random ordering has an expected area of one half. A completely reversed ranking scores zero. Below one half can signal a flipped score direction, but don't flip a model just because of your test set. And an A U C of zero point eight does not mean eighty percent of predictions are correct. + +## 05:34.5 — Same ordering. Same AUC. Different numbers. + +Now square every score. The numbers change, but the order doesn't, so the R O C curve and its area stay at seven ninths. A U C can't tell you which set of numbers makes better probabilities. A fixed cutoff has to move too: zero point six five becomes its square, about zero point four two. Check calibration separately, and pick the threshold from the costs of mistakes. + +## 06:02.5 — A small false-positive rate can mean many alarms + +Suppose a threshold catches eighty percent of frauds and flags ten percent of legitimate payments. With a hundred of each, that's eighty true alarms and ten false ones: about eighty nine percent precision. Now take ten frauds among nine hundred ninety legitimate payments. The same rates give eight true alarms and ninety nine false ones: only seven and a half percent precision. The R O C point never moved: eighty percent, ten percent. Precision did, because prevalence, the Greek letter pi, changes what the flagged pile is made of. + +## 06:41.5 — Precision asks about the flagged pile + +Back to our six payments. Recall is just another name for the true positive rate. Precision asks a different question: of the payments we flagged, how many are really fraud? A precision recall curve plots recall across and precision up, as we sweep the same threshold. Each fraud we find raises recall; each legitimate payment we flag drags precision down. Before anything is flagged, precision is undefined, so plots usually start the curve at one. + +## 07:15.0 — Reward precision when a new positive is found + +A common summary is average precision, or A P. Each new fraud raises recall by one third. At those moments, precision is one, then two thirds, then three quarters. Weight each by its recall step and add them up. Twenty-nine over thirty-six, about zero point eight one. Same predictions as our R O C area of seven ninths, but a different question, and a different number. + +## 07:44.5 — PR-AUC and AP are not interchangeable labels + +Average precision sums recall steps times precision: zero point eight oh six. Join the raw precision recall points with straight lines and take trapezoids, and you get zero point seven six four. Those straight lines aren't even a valid interpolation in precision recall space. Other benchmarks use an interpolated envelope: at each recall, the best precision at that recall or higher. Here it lifts the middle step from two thirds to three quarters. And on large samples, a random ranker's baseline precision is the prevalence: one half here, only because half our payments are fraud. Always name the convention. + +## 08:30.0 — Partial AUC: only the region you operate in + +Say you can tolerate almost no false alarms. Partial A U C keeps only false positive rates up to zero point one. Here the raw area is one thirtieth, out of a possible one tenth. Scikit-learn's max F P R setting rescales that so one half is chance and one is perfect: about zero point six five. Always state the range, and whether it was rescaled. + +## 08:58.5 — Multiclass AUC needs a decomposition and an average + +Now suppose a second model sorts confirmed frauds into three types. One versus rest turns each class into its own binary problem, with its own R O C area; one versus one compares pairs of classes instead. The macro average weights the classes equally: about zero point seven eight. Weighting by class size gives zero point nine one, because the biggest class is the easiest. Same model, two honest numbers with different priorities. And when an example can carry several labels, a micro average pools every example label pair into one curve; that is not an average of areas. + +## 09:42.0 — Same letters, different curves + +Outside classification, the same letters show up on different curves. In search, average precision rewards relevant results near the top, and mean average precision averages it over queries. With three relevant results, at ranks one, three, and four, it's our twenty-nine over thirty-six again. Object detection first matches each predicted box to a true one by their overlap, called I O U. These two boxes overlap by two thirds: a match at one half, a miss at three quarters. Only then does it build precision recall curves, so Coco's A P is not an R O C area. Survival models compute an A U C at each time horizon t. A machine that failed by t is a case, one still running at t is a control, and one last seen before t is unknown, not a negative. And area under a learning curve measures performance across a training budget, not ranking at all. + +## 10:42.5 — Keep the curve, the data, and the decision separate + +In practice, feed the metric scores, not hard labels, and make sure larger means more likely positive. Evaluate on held-out data, and report uncertainty that matches how you sampled. Look at the curve and at important subgroups, not just one number. Then choose a threshold from your costs and your prevalence. No kind of A U C chooses that for you. + +## 11:08.5 — Ask five questions when someone says “AUC” + +Same six payments, two curves, and two different numbers: seven ninths, and twenty-nine over thirty-six. Neither is wrong; they answer different questions. R O C area asks whether frauds outrank legitimate payments. Precision recall asks how clean the flagged pile stays as recall grows. So when you hear A U C, ask: which curve, which positive class, which population, which range, and which average? diff --git a/videos/auc-ml/tsconfig.json b/videos/auc-ml/tsconfig.json new file mode 100644 index 0000000..c8e5eef --- /dev/null +++ b/videos/auc-ml/tsconfig.json @@ -0,0 +1 @@ +{ "extends": "../../tsconfig.base.json", "compilerOptions": { "noEmit": true }, "include": ["src"] } diff --git a/videos/auc-ml/verify_math.py b/videos/auc-ml/verify_math.py new file mode 100644 index 0000000..3dd6b44 --- /dev/null +++ b/videos/auc-ml/verify_math.py @@ -0,0 +1,37 @@ +from fractions import Fraction as F +import json +from pathlib import Path +scores=[F(9,10),F(8,10),F(7,10),F(6,10),F(4,10),F(1,10)] +y=[1,0,1,1,0,0] +pos=[s for s,l in zip(scores,y) if l];neg=[s for s,l in zip(scores,y) if not l] +def pair_auc(p,n):return sum(F(a>b)+F(1,2)*F(a==b) for a in p for b in n)/(len(p)*len(n)) +roc=[(F(0),F(0))];pr=[(F(0),F(1))];tp=fp=0;ap=F(0) +for k,l in enumerate(y,1): + tp+=l;fp+=1-l;roc.append((F(fp,3),F(tp,3)));pr.append((F(tp,3),F(tp,k))) + if l:ap+=F(1,3)*F(tp,k) +def trap(points):return sum((x1-x0)*(y0+y1)/2 for (x0,y0),(x1,y1) in zip(points,points[1:])) +chosen=[i for i,s in enumerate(scores) if s>=F(65,100)] +assert chosen==[0,1,2] and sum(y[i] for i in chosen)==2 +assert F(80,90)==F(8,9) and F(8,8+99)==F(8,107) +assert pair_auc(pos,neg)==trap(roc)==F(7,9) +assert ap==F(29,36) +assert trap(pr)==F(55,72) +assert pair_auc([s*s for s in pos],[s*s for s in neg])==F(7,9) +assert pair_auc([-s for s in pos],[-s for s in neg])==F(2,9) +assert pair_auc([F(1,2)],[F(1,2)])==F(1,2) +assert F(1,10)*F(1,3)==F(1,30) +assert sum([F(95,100),F(80,100),F(60,100)])/3==F(47,60) +assert F(95,100)*F(80,100)+F(80,100)*F(15,100)+F(60,100)*F(5,100)==F(91,100) +assert F(8000,20000-8000)==F(2,3) +# Standardized partial AUC (McClish, as scikit-learn max_fpr): 0.5*(1+(A-min)/(max-min)), min = a^2/2. +a=F(1,10);pa=F(1,30);std=F(1,2)*(1+(pa-a*a/2)/(a-a*a/2)) +assert abs(float(std)-0.649)<0.0005 +# Interpolated precision envelope used on screen: max precision at recall >= r. +env={r:max(p for rr,p in pr[1:] if rr>=r) for r in (F(1,3),F(2,3),F(1))} +assert env=={F(1,3):F(1),F(2,3):F(3,4),F(1):F(3,4)} +# Retrieval callback: relevant at ranks 1,3,4 of 6 is the same AP. +assert ap==F(1,3)*(F(1,1)+F(2,3)+F(3,4)) +# Imbalance: identical ROC point in both populations. +assert (F(80,100),F(10,100))==(F(8,10),F(99,990)) +result={'roc_auc':str(trap(roc)),'pair_auc':str(pair_auc(pos,neg)),'average_precision':str(ap),'linear_pr_auc':str(trap(pr)),'roc':[[float(x),float(y)] for x,y in roc],'pr':[[float(x),float(y)] for x,y in pr],'checks':'threshold counts, pair equality, AP, trapezoids, ties, reversal, monotone transform, partial area, averaging, IoU, standardized partial AUC, interpolated envelope, retrieval AP, prevalence-invariant ROC point'} +out=Path(__file__).parent/'out';out.mkdir(exist_ok=True);(out/'math-verification.json').write_text(json.dumps(result,indent=2)+'\n');print(json.dumps(result,indent=2))