#!/usr/bin/env node 'use strict'; /** * fit-policy-adjudication — MEASURE the production fitting policy, do not change it. * * The artifact now has an identity. The open question is whether the PROCEDURE * that produces tomorrow's curve deserves continuous trust. * * Candidates are fitted SIDE-EFFECT-FREE as of the production `fit_as_of`, and * compared both on their mapping and on point-in-time out-of-sample score. * No policy is written anywhere by this script. */ require('dotenv').config({ quiet: true }); const { createClient } = require('@supabase/supabase-js'); const crypto = require('crypto'); const cal = require('../src/services/model/calibration'); const calSvc = require('../src/services/model/calibrationService'); const ERA = 'engine1@2026-08-07-fullwindow'; const FIT_AS_OF = process.env.FIT_AS_OF || '2026-09-02'; const PROBES = [0.50, 0.55, 0.60, 0.65, 0.70, 0.75, 0.79]; const r3 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 1000) / 1000); const r5 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 100000) / 100000); const dg = (o) => crypto.createHash('sha256').update(JSON.stringify(o)).digest('hex').slice(0, 16); const brier = (ps, ys) => (ps.length ? ps.reduce((s, p, i) => s + (p - ys[i]) ** 2, 0) / ps.length : null); function logloss(ps, ys) { const E = 1e-12; let s = 0; for (let i = 0; i < ps.length; i++) { const p = Math.min(1 - E, Math.max(E, ps[i])); s += -(ys[i] * Math.log(p) + (1 - ys[i]) * Math.log(1 - p)); } return ps.length ? s / ps.length : null; } function ece(ps, ys, bins = 10) { const a = Array.from({ length: bins }, () => ({ n: 0, sp: 0, sy: 0 })); for (let i = 0; i < ps.length; i++) { const b = Math.min(bins - 1, Math.floor(ps[i] * bins)); a[b].n++; a[b].sp += ps[i]; a[b].sy += ys[i]; } let e = 0; for (const b of a) if (b.n) e += (b.n / ps.length) * Math.abs(b.sp / b.n - b.sy / b.n); return e; } function pairedCI(a, b, ys, iters = 2000, seed = 11) { let s = seed >>> 0; const rnd = () => { s = (s * 1664525 + 1013904223) >>> 0; return s / 4294967296; }; const n = ys.length, out = []; for (let it = 0; it < iters; it++) { let sa = 0, sb = 0; for (let i = 0; i < n; i++) { const j = Math.floor(rnd() * n); sa += (a[j] - ys[j]) ** 2; sb += (b[j] - ys[j]) ** 2; } out.push(sa / n - sb / n); } out.sort((x, y) => x - y); return [out[Math.floor(iters * 0.025)], out[Math.floor(iters * 0.975)]]; } (async () => { const sb = createClient(process.env.SUPABASE_URL, process.env.SUPABASE_SERVICE_KEY, { auth: { persistSession: false } }); // The production walk, verbatim — no model filter, strictly before the cutoff. const raw = await calSvc.loadSettledRows(sb, { sport: 'mlb', stat: 'hits', before: FIT_AS_OF }); // ERA COMES FROM THE WALK, NOT FROM AN `.in('id', [...])` BACKFILL. // A chunked id filter is a URL, and 500 UUIDs is an 18,000-character request // the fetch layer rejects — chunks return nothing and the rows read as // "unknown era". Measured here first time round: it reported the superseded // era as 2.1% of the fit when the true share is over half. const { paginate } = require('../src/utils/safePaginate'); const withEra = await paginate( () => sb.from('ledger_entries') .select('id, p_win, outcome, game_date, quarantine_reason, model_version') .eq('sport', 'mlb').is('user_id', null).eq('stat', 'hits') .in('outcome', ['hit', 'miss']).not('p_win', 'is', null) .lt('game_date', FIT_AS_OF), { key: 'id', pageSize: 1000, label: 'fit-policy era walk' }); if (withEra.length !== raw.length) { throw new Error(`era walk (${withEra.length}) disagrees with the production walk (${raw.length})`); } const all = withEra .filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book')) .map((r) => ({ p: Number(r.p_win), won: r.outcome === 'hit' ? 1 : 0, date: String(r.game_date), era: r.model_version || null })) .filter((r) => Number.isFinite(r.p)) .sort((a, b) => (a.date < b.date ? -1 : a.date > b.date ? 1 : 0)); const eraCounts = {}; for (const r of all) eraCounts[r.era || 'unknown'] = (eraCounts[r.era || 'unknown'] || 0) + 1; // ── STEP 16 — THE CURRENT POLICY, EXACTLY ────────────────────────────── const cut = Math.floor(all.length * 0.65); const curFit = all.slice(0, cut); const curHeld = all.slice(cut); const curEraMix = {}; for (const r of curFit) curEraMix[r.era || 'unknown'] = (curEraMix[r.era || 'unknown'] || 0) + 1; console.log(JSON.stringify({ section: 'CURRENT_POLICY', fit_as_of: FIT_AS_OF, query: { table: 'ledger_entries', filters: ["sport=mlb", "user_id IS NULL", "stat=hits", "outcome IN (hit,miss)", "p_win NOT NULL", `game_date < ${FIT_AS_OF}`], model_version_filter: 'NONE — the fit pools every model era' }, rows_walked: raw.length, rows_clean: all.length, era_counts: eraCounts, fit_fraction: 0.65, fit_n: curFit.length, withheld_n: curHeld.length, fit_first_date: curFit[0].date, fit_last_date: curFit[curFit.length - 1].date, withheld_first_date: curHeld[0].date, withheld_last_date: curHeld[curHeld.length - 1].date, fit_era_mix: curEraMix, superseded_era_share_of_fit: r3((curEraMix['engine1@2026-07-20'] || 0) / curFit.length), refit_cadence: 'once per snapshot run (probabilityContractService.build), cutoff = todayEt()', object: 'A — a continuously refitted estimator PROCEDURE, not a frozen artifact', }, null, 1)); // ── STEPS 18/19 — CANDIDATE POLICIES, side-effect-free ──────────────── const eraRows = all.filter((r) => r.era === ERA); const policies = []; const mk = (id, desc, fitRows, heldRows) => { const map = cal.fitIsotonic(fitRows, { minTotal: 200 }); return { id, desc, fit_n: fitRows.length, held_n: heldRows.length, fit_last_date: fitRows.length ? fitRows[fitRows.length - 1].date : null, map, knot_count: map ? map.length : null, knot_digest: map ? dg(map) : null, probes: Object.fromEntries(PROBES.map((p) => [p, map ? r3(cal.applyIsotonic(map, p)) : null])) }; }; policies.push(mk('A_CURRENT_65_35_ALL_ERAS', 'production today: 65% of ALL eras pooled', curFit, curHeld)); const eCut = Math.floor(eraRows.length * 0.65); policies.push(mk('B_ERA_FILTERED_65_35', 'same split, current model era only', eraRows.slice(0, eCut), eraRows.slice(eCut))); // C: expanding train, fixed RECENT holdout (era-filtered) — holdout by DATE, not fraction const eraDates = [...new Set(eraRows.map((r) => r.date))].sort(); const holdFrom = eraDates[Math.max(0, eraDates.length - 4)]; // last 4 dates held policies.push(mk('C_ERA_EXPANDING_FIXED_RECENT_HOLDOUT', 'current era, all but the last 4 settled dates', eraRows.filter((r) => r.date < holdFrom), eraRows.filter((r) => r.date >= holdFrom))); // D: era-filtered, ALL rows fitted (no withhold at all) policies.push(mk('D_ERA_ALL_ROWS_NO_WITHHOLD', 'current era, every settled row fitted', eraRows, [])); console.log(JSON.stringify({ section: 'CANDIDATE_MAPS', era_rows: eraRows.length, holdout_from_date_for_C: holdFrom, policies: policies.map(({ map, ...rest }) => rest) }, null, 1)); const base = policies[0]; console.log(JSON.stringify({ section: 'MAP_DELTAS_VS_PRODUCTION', deltas: policies.slice(1).map((p) => ({ id: p.id, per_probe: Object.fromEntries(PROBES.map((x) => [x, (p.probes[x] == null || base.probes[x] == null) ? null : r3(p.probes[x] - base.probes[x])])), max_abs: r3(Math.max(...PROBES.map((x) => Math.abs((p.probes[x] ?? 0) - (base.probes[x] ?? 0))))) })) }, null, 1)); // ── OUT-OF-SAMPLE, POINT IN TIME ─────────────────────────────────────── // Every policy is fitted on evidence strictly before EVAL_FROM and scored on // current-era rows at/after it. Same rows for every policy. const EVAL_FROM = eraDates[Math.max(0, eraDates.length - 4)]; const evalRows = eraRows.filter((r) => r.date >= EVAL_FROM); const trainAll = all.filter((r) => r.date < EVAL_FROM); const trainEra = eraRows.filter((r) => r.date < EVAL_FROM); const oos = []; const fitFor = (id) => { if (id === 'A_CURRENT_65_35_ALL_ERAS') return trainAll.slice(0, Math.floor(trainAll.length * 0.65)); if (id === 'B_ERA_FILTERED_65_35') return trainEra.slice(0, Math.floor(trainEra.length * 0.65)); if (id === 'C_ERA_EXPANDING_FIXED_RECENT_HOLDOUT') return trainEra; return trainEra; }; const ys = evalRows.map((r) => r.won); const rawPs = evalRows.map((r) => r.p); const inSupport = (p) => p >= 0.50 && p < 0.80; const covIdx = evalRows.map((r, i) => (inSupport(r.p) ? i : -1)).filter((i) => i >= 0); const covY = covIdx.map((i) => ys[i]); const covRaw = covIdx.map((i) => rawPs[i]); for (const id of ['A_CURRENT_65_35_ALL_ERAS', 'B_ERA_FILTERED_65_35', 'C_ERA_EXPANDING_FIXED_RECENT_HOLDOUT', 'D_ERA_ALL_ROWS_NO_WITHHOLD']) { const fr = fitFor(id); const m = cal.fitIsotonic(fr, { minTotal: 200 }); if (!m) { oos.push({ id, refused: true, fit_n: fr.length }); continue; } const served = covIdx.map((i) => cal.applyIsotonic(m, rawPs[i]) ?? rawPs[i]); const ci = pairedCI(served, covRaw, covY); oos.push({ id, fit_n: fr.length, fit_last_date: fr[fr.length - 1].date, knot_digest: dg(m), eval_n: covY.length, eval_from: EVAL_FROM, brier: r5(brier(served, covY)), logloss: r5(logloss(served, covY)), ece: r5(ece(served, covY)), delta_vs_raw: r5(brier(served, covY) - brier(covRaw, covY)), ci95: [r5(ci[0]), r5(ci[1])], probes: Object.fromEntries(PROBES.map((p) => [p, r3(cal.applyIsotonic(m, p))])) }); } console.log(JSON.stringify({ section: 'OUT_OF_SAMPLE_POINT_IN_TIME', eval_from: EVAL_FROM, eval_rows_in_support: covY.length, brier_raw_on_same_rows: r5(brier(covRaw, covY)), policies: oos }, null, 1)); process.exit(0); })().catch((e) => { console.error(e); process.exit(1); });