Files
vyndr/scripts/fit-policy-adjudication.js
builtbykev 556d186ff1 One evaluator for the shadow, so the probe cannot report a mode the pipeline is not in
"Is the shadow effective?" was only answerable by waiting for a snapshot to
write a row. That leaves a blind spot with real cost: a variable SET IN COOLIFY
BUT NOT YET APPLIED to the running process is indistinguishable from an unset
one, and the runtime probe already proves the distinction matters — code_sha
22cf51c with started_at 01:45:44Z means anything set after that is not in this
process's environment.

probabilityContract.shadowState() is now the single evaluator. snapshotService
calls it and the protected status probe calls it, and a test asserts NEITHER
reads process.env directly — the same rule that keeps lineage_write_mode honest.
Reading the env in two places is how a status page and a gate come to disagree.

Strict by construction: only the exact string '1' enables it. 'true', 'yes',
'on', '01', ' 1 ' and '' are all OFF, because a loose parse turns a typo into an
activation. `configuration_source` separates an unset variable from one
explicitly set to '0', and `live_serving` is reported as its own switch so the
shadow can never be read as implying serving.

No behaviour changes. The shadow still defaults OFF, CALIBRATION_DEPLOYED is
still [], and served fields are untouched.

Frontend byte-identical to the last green build (git reports zero changes under
web/), so the build from 22cf51c stands.

Suite 401/401, 5,597 passed, 4 skipped.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CQJeAG8vcDoL5zkiaJyVb8
2026-09-02 22:53:21 -04:00

167 lines
9.9 KiB
JavaScript

#!/usr/bin/env node
'use strict';
/**
* fit-policy-adjudication — MEASURE the production fitting policy, do not change it.
*
* The artifact now has an identity. The open question is whether the PROCEDURE
* that produces tomorrow's curve deserves continuous trust.
*
* Candidates are fitted SIDE-EFFECT-FREE as of the production `fit_as_of`, and
* compared both on their mapping and on point-in-time out-of-sample score.
* No policy is written anywhere by this script.
*/
require('dotenv').config({ quiet: true });
const { createClient } = require('@supabase/supabase-js');
const crypto = require('crypto');
const cal = require('../src/services/model/calibration');
const calSvc = require('../src/services/model/calibrationService');
const ERA = 'engine1@2026-08-07-fullwindow';
const FIT_AS_OF = process.env.FIT_AS_OF || '2026-09-02';
const PROBES = [0.50, 0.55, 0.60, 0.65, 0.70, 0.75, 0.79];
const r3 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 1000) / 1000);
const r5 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 100000) / 100000);
const dg = (o) => crypto.createHash('sha256').update(JSON.stringify(o)).digest('hex').slice(0, 16);
const brier = (ps, ys) => (ps.length ? ps.reduce((s, p, i) => s + (p - ys[i]) ** 2, 0) / ps.length : null);
function logloss(ps, ys) { const E = 1e-12; let s = 0;
for (let i = 0; i < ps.length; i++) { const p = Math.min(1 - E, Math.max(E, ps[i]));
s += -(ys[i] * Math.log(p) + (1 - ys[i]) * Math.log(1 - p)); } return ps.length ? s / ps.length : null; }
function ece(ps, ys, bins = 10) { const a = Array.from({ length: bins }, () => ({ n: 0, sp: 0, sy: 0 }));
for (let i = 0; i < ps.length; i++) { const b = Math.min(bins - 1, Math.floor(ps[i] * bins));
a[b].n++; a[b].sp += ps[i]; a[b].sy += ys[i]; }
let e = 0; for (const b of a) if (b.n) e += (b.n / ps.length) * Math.abs(b.sp / b.n - b.sy / b.n); return e; }
function pairedCI(a, b, ys, iters = 2000, seed = 11) { let s = seed >>> 0;
const rnd = () => { s = (s * 1664525 + 1013904223) >>> 0; return s / 4294967296; };
const n = ys.length, out = [];
for (let it = 0; it < iters; it++) { let sa = 0, sb = 0;
for (let i = 0; i < n; i++) { const j = Math.floor(rnd() * n); sa += (a[j] - ys[j]) ** 2; sb += (b[j] - ys[j]) ** 2; }
out.push(sa / n - sb / n); }
out.sort((x, y) => x - y); return [out[Math.floor(iters * 0.025)], out[Math.floor(iters * 0.975)]]; }
(async () => {
const sb = createClient(process.env.SUPABASE_URL, process.env.SUPABASE_SERVICE_KEY,
{ auth: { persistSession: false } });
// The production walk, verbatim — no model filter, strictly before the cutoff.
const raw = await calSvc.loadSettledRows(sb, { sport: 'mlb', stat: 'hits', before: FIT_AS_OF });
// ERA COMES FROM THE WALK, NOT FROM AN `.in('id', [...])` BACKFILL.
// A chunked id filter is a URL, and 500 UUIDs is an 18,000-character request
// the fetch layer rejects — chunks return nothing and the rows read as
// "unknown era". Measured here first time round: it reported the superseded
// era as 2.1% of the fit when the true share is over half.
const { paginate } = require('../src/utils/safePaginate');
const withEra = await paginate(
() => sb.from('ledger_entries')
.select('id, p_win, outcome, game_date, quarantine_reason, model_version')
.eq('sport', 'mlb').is('user_id', null).eq('stat', 'hits')
.in('outcome', ['hit', 'miss']).not('p_win', 'is', null)
.lt('game_date', FIT_AS_OF),
{ key: 'id', pageSize: 1000, label: 'fit-policy era walk' });
if (withEra.length !== raw.length) {
throw new Error(`era walk (${withEra.length}) disagrees with the production walk (${raw.length})`);
}
const all = withEra
.filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book'))
.map((r) => ({ p: Number(r.p_win), won: r.outcome === 'hit' ? 1 : 0,
date: String(r.game_date), era: r.model_version || null }))
.filter((r) => Number.isFinite(r.p))
.sort((a, b) => (a.date < b.date ? -1 : a.date > b.date ? 1 : 0));
const eraCounts = {};
for (const r of all) eraCounts[r.era || 'unknown'] = (eraCounts[r.era || 'unknown'] || 0) + 1;
// ── STEP 16 — THE CURRENT POLICY, EXACTLY ──────────────────────────────
const cut = Math.floor(all.length * 0.65);
const curFit = all.slice(0, cut);
const curHeld = all.slice(cut);
const curEraMix = {};
for (const r of curFit) curEraMix[r.era || 'unknown'] = (curEraMix[r.era || 'unknown'] || 0) + 1;
console.log(JSON.stringify({ section: 'CURRENT_POLICY', fit_as_of: FIT_AS_OF,
query: { table: 'ledger_entries', filters: ["sport=mlb", "user_id IS NULL", "stat=hits",
"outcome IN (hit,miss)", "p_win NOT NULL", `game_date < ${FIT_AS_OF}`],
model_version_filter: 'NONE — the fit pools every model era' },
rows_walked: raw.length, rows_clean: all.length, era_counts: eraCounts,
fit_fraction: 0.65, fit_n: curFit.length, withheld_n: curHeld.length,
fit_first_date: curFit[0].date, fit_last_date: curFit[curFit.length - 1].date,
withheld_first_date: curHeld[0].date, withheld_last_date: curHeld[curHeld.length - 1].date,
fit_era_mix: curEraMix,
superseded_era_share_of_fit: r3((curEraMix['engine1@2026-07-20'] || 0) / curFit.length),
refit_cadence: 'once per snapshot run (probabilityContractService.build), cutoff = todayEt()',
object: 'A — a continuously refitted estimator PROCEDURE, not a frozen artifact',
}, null, 1));
// ── STEPS 18/19 — CANDIDATE POLICIES, side-effect-free ────────────────
const eraRows = all.filter((r) => r.era === ERA);
const policies = [];
const mk = (id, desc, fitRows, heldRows) => {
const map = cal.fitIsotonic(fitRows, { minTotal: 200 });
return { id, desc, fit_n: fitRows.length, held_n: heldRows.length,
fit_last_date: fitRows.length ? fitRows[fitRows.length - 1].date : null,
map, knot_count: map ? map.length : null, knot_digest: map ? dg(map) : null,
probes: Object.fromEntries(PROBES.map((p) => [p, map ? r3(cal.applyIsotonic(map, p)) : null])) };
};
policies.push(mk('A_CURRENT_65_35_ALL_ERAS', 'production today: 65% of ALL eras pooled', curFit, curHeld));
const eCut = Math.floor(eraRows.length * 0.65);
policies.push(mk('B_ERA_FILTERED_65_35', 'same split, current model era only', eraRows.slice(0, eCut), eraRows.slice(eCut)));
// C: expanding train, fixed RECENT holdout (era-filtered) — holdout by DATE, not fraction
const eraDates = [...new Set(eraRows.map((r) => r.date))].sort();
const holdFrom = eraDates[Math.max(0, eraDates.length - 4)]; // last 4 dates held
policies.push(mk('C_ERA_EXPANDING_FIXED_RECENT_HOLDOUT', 'current era, all but the last 4 settled dates',
eraRows.filter((r) => r.date < holdFrom), eraRows.filter((r) => r.date >= holdFrom)));
// D: era-filtered, ALL rows fitted (no withhold at all)
policies.push(mk('D_ERA_ALL_ROWS_NO_WITHHOLD', 'current era, every settled row fitted', eraRows, []));
console.log(JSON.stringify({ section: 'CANDIDATE_MAPS', era_rows: eraRows.length,
holdout_from_date_for_C: holdFrom,
policies: policies.map(({ map, ...rest }) => rest) }, null, 1));
const base = policies[0];
console.log(JSON.stringify({ section: 'MAP_DELTAS_VS_PRODUCTION',
deltas: policies.slice(1).map((p) => ({ id: p.id,
per_probe: Object.fromEntries(PROBES.map((x) => [x,
(p.probes[x] == null || base.probes[x] == null) ? null : r3(p.probes[x] - base.probes[x])])),
max_abs: r3(Math.max(...PROBES.map((x) => Math.abs((p.probes[x] ?? 0) - (base.probes[x] ?? 0))))) })) }, null, 1));
// ── OUT-OF-SAMPLE, POINT IN TIME ───────────────────────────────────────
// Every policy is fitted on evidence strictly before EVAL_FROM and scored on
// current-era rows at/after it. Same rows for every policy.
const EVAL_FROM = eraDates[Math.max(0, eraDates.length - 4)];
const evalRows = eraRows.filter((r) => r.date >= EVAL_FROM);
const trainAll = all.filter((r) => r.date < EVAL_FROM);
const trainEra = eraRows.filter((r) => r.date < EVAL_FROM);
const oos = [];
const fitFor = (id) => {
if (id === 'A_CURRENT_65_35_ALL_ERAS') return trainAll.slice(0, Math.floor(trainAll.length * 0.65));
if (id === 'B_ERA_FILTERED_65_35') return trainEra.slice(0, Math.floor(trainEra.length * 0.65));
if (id === 'C_ERA_EXPANDING_FIXED_RECENT_HOLDOUT') return trainEra;
return trainEra;
};
const ys = evalRows.map((r) => r.won);
const rawPs = evalRows.map((r) => r.p);
const inSupport = (p) => p >= 0.50 && p < 0.80;
const covIdx = evalRows.map((r, i) => (inSupport(r.p) ? i : -1)).filter((i) => i >= 0);
const covY = covIdx.map((i) => ys[i]);
const covRaw = covIdx.map((i) => rawPs[i]);
for (const id of ['A_CURRENT_65_35_ALL_ERAS', 'B_ERA_FILTERED_65_35',
'C_ERA_EXPANDING_FIXED_RECENT_HOLDOUT', 'D_ERA_ALL_ROWS_NO_WITHHOLD']) {
const fr = fitFor(id);
const m = cal.fitIsotonic(fr, { minTotal: 200 });
if (!m) { oos.push({ id, refused: true, fit_n: fr.length }); continue; }
const served = covIdx.map((i) => cal.applyIsotonic(m, rawPs[i]) ?? rawPs[i]);
const ci = pairedCI(served, covRaw, covY);
oos.push({ id, fit_n: fr.length, fit_last_date: fr[fr.length - 1].date,
knot_digest: dg(m),
eval_n: covY.length, eval_from: EVAL_FROM,
brier: r5(brier(served, covY)), logloss: r5(logloss(served, covY)), ece: r5(ece(served, covY)),
delta_vs_raw: r5(brier(served, covY) - brier(covRaw, covY)), ci95: [r5(ci[0]), r5(ci[1])],
probes: Object.fromEntries(PROBES.map((p) => [p, r3(cal.applyIsotonic(m, p))])) });
}
console.log(JSON.stringify({ section: 'OUT_OF_SAMPLE_POINT_IN_TIME',
eval_from: EVAL_FROM, eval_rows_in_support: covY.length,
brier_raw_on_same_rows: r5(brier(covRaw, covY)), policies: oos }, null, 1));
process.exit(0);
})().catch((e) => { console.error(e); process.exit(1); });