be8e16aca9
The last release detected the violation and then served the certified state anyway. A validator that changes nothing is decoration, so `servable:false` is now load-bearing: an artifact that fails its policy returns ARTIFACT_POLICY_BLOCKED with no number, and every probability-derived claim goes with it. The gate sits inside the resolution, not beside the flag that turns the shadow on, so no environment variable can reach past it — a test asserts `resolve` never reads process.env at all. Shadow and live consume the SAME decision, differing only in which promotion stage they demand. Era mismatch still resolves to VERSION_MISMATCH rather than the new state. "This artifact belongs to a different forecaster" is more precise than "policy blocked", and the existing state already says it exactly. THE REPAIR. `currentEraSource` filters on model_version in the QUERY, taking the era from config/modelVersion so the query, the artifact and the validator all read one identity. Measured on the actual fitted set, not a second count: 6,069 current-era rows, 0 wrong-era. The procedure was then certified on current-era rows ONLY — four walk-forward folds, training strictly before each evaluation block, 0 future rows in train on every fold. All four improve; pooled n=3,108 gives Brier 0.24701 -> 0.24323, delta -0.00378, CI [-0.00619,-0.00147] excluding zero; ECE falls in every fold. Mapping spread inside support is 0.001-0.018. The prior mixed-era certification did not substitute for this. Policy B selected. A (era-filtered 65/35) and B (all current-era) are statistically indistinguishable, A-B = +0.0001 CI [-0.00029,+0.00048], but B has the better ECE (0.0064 vs 0.0109) and the holdout existed to certify the PROCEDURE — it is not permanently withheld from the artifact that ships. withheld_from_fit is 0. FROZEN. `mlb-hits-isotonic@2026-09-03`: 6,069 rows, training_cutoff 2026-09-01 (distinct from fit_as_of 2026-09-03 — the newest observation admitted is not the eligibility bound), 12 knots, source_digest 25919c16…, knot_digest 5ae940ea…, served_curve_digest c24a9dc5…, 8 curve steps, 924 bytes, committed as JSON. The runtime no longer fits. It loads. A test greps the service for fitIsotonic, fromLedger and loadRows and requires all three absent, because the old behaviour meant a user's number could move with no version, no review and no rollback, and a past Read could not be reconstructed because its curve no longer existed. New settled outcomes are forward evidence now; they cannot touch this curve. Independent reconstruction from the declared training contract alone — fresh read, fresh digest, fresh fit — reproduces every digest and the curve byte for byte. Calling the builder twice would only have proven the builder deterministic. Promotion is a frozen source constant. A snapshot cannot promote, a settlement cannot promote, a successful fit cannot promote, and dropping a file into the artifacts directory promotes nothing. Stage is APPROVED_FOR_SHADOW; live is explicitly false. Two coverage holes found by their own teeth. The promotion guard could be deleted with every test still green, because the promoted file naturally agrees with itself — extracted as `acceptFile` and tested on the case `load()` cannot reach. And `validate(null)` returned no `servable` field at all, which is falsy at a call site and so would have read as correct while asserting nothing. Shadow OFF. Live OFF. CALIBRATION_DEPLOYED []. No frontend change. Suite 404/404, 5,634 passed, 4 skipped. Teeth 26/26 + 10/10 + 23/23. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CQJeAG8vcDoL5zkiaJyVb8
153 lines
8.9 KiB
JavaScript
153 lines
8.9 KiB
JavaScript
#!/usr/bin/env node
|
|
'use strict';
|
|
/**
|
|
* certify-current-era-procedure — prove the REFIT PROCEDURE on current-era rows only.
|
|
*
|
|
* The prior certification pooled model eras. It does not substitute for this:
|
|
* a procedure restricted to one forecaster has to earn its own proof, on that
|
|
* forecaster's observations, walk-forward.
|
|
*
|
|
* SUPABASE_URL=... SUPABASE_SERVICE_KEY=... node scripts/certify-current-era-procedure.js
|
|
*/
|
|
require('dotenv').config({ quiet: true });
|
|
const { createClient } = require('@supabase/supabase-js');
|
|
const cal = require('../src/services/model/calibration');
|
|
const src = require('../src/services/model/currentEraSource');
|
|
const { MODEL_VERSION } = require('../src/config/modelVersion');
|
|
|
|
const SPORT = 'mlb';
|
|
const STAT = 'hits';
|
|
const SUPPORT = [0.50, 0.80];
|
|
const PROBES = [0.50, 0.55, 0.60, 0.65, 0.70, 0.75, 0.79];
|
|
const MIN_FIT = 200;
|
|
const FOLD_DATES = Number(process.env.FOLD_DATES || 3); // eval block width
|
|
const FOLDS = Number(process.env.FOLDS || 4);
|
|
|
|
const r3 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 1000) / 1000);
|
|
const r5 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 100000) / 100000);
|
|
const inSupport = (p) => p >= SUPPORT[0] && p < SUPPORT[1];
|
|
|
|
const brier = (ps, ys) => (ps.length ? ps.reduce((s, p, i) => s + (p - ys[i]) ** 2, 0) / ps.length : null);
|
|
function logloss(ps, ys) { const E = 1e-12; let s = 0;
|
|
for (let i = 0; i < ps.length; i++) { const p = Math.min(1 - E, Math.max(E, ps[i]));
|
|
s += -(ys[i] * Math.log(p) + (1 - ys[i]) * Math.log(1 - p)); } return ps.length ? s / ps.length : null; }
|
|
function ece(ps, ys, bins = 10) { const a = Array.from({ length: bins }, () => ({ n: 0, sp: 0, sy: 0 }));
|
|
for (let i = 0; i < ps.length; i++) { const b = Math.min(bins - 1, Math.floor(ps[i] * bins));
|
|
a[b].n++; a[b].sp += ps[i]; a[b].sy += ys[i]; }
|
|
let e = 0; for (const b of a) if (b.n) e += (b.n / ps.length) * Math.abs(b.sp / b.n - b.sy / b.n); return e; }
|
|
function wilson(k, n, z = 1.96) { if (!n) return null; const p = k / n, d = 1 + z * z / n;
|
|
const c = (p + z * z / (2 * n)) / d, h = (z * Math.sqrt(p * (1 - p) / n + z * z / (4 * n * n))) / d;
|
|
return [Math.max(0, c - h), Math.min(1, c + h)]; }
|
|
function pairedCI(a, b, ys, iters = 2000, seed = 17) { let s = seed >>> 0;
|
|
const rnd = () => { s = (s * 1664525 + 1013904223) >>> 0; return s / 4294967296; };
|
|
const n = ys.length, out = [];
|
|
for (let it = 0; it < iters; it++) { let sa = 0, sb = 0;
|
|
for (let i = 0; i < n; i++) { const j = Math.floor(rnd() * n); sa += (a[j] - ys[j]) ** 2; sb += (b[j] - ys[j]) ** 2; }
|
|
out.push(sa / n - sb / n); }
|
|
out.sort((x, y) => x - y); return [out[Math.floor(iters * 0.025)], out[Math.floor(iters * 0.975)]]; }
|
|
const summ = (a) => { if (!a.length) return { support: 0 };
|
|
const s = [...a].sort((x, y) => x - y); const q = (f) => s[Math.min(s.length - 1, Math.floor(s.length * f))];
|
|
return { support: s.length, median: r3(q(0.5)), min: r3(s[0]), max: r3(s[s.length - 1]),
|
|
iqr: r3(q(0.75) - q(0.25)), spread: r3(s[s.length - 1] - s[0]) }; };
|
|
|
|
(async () => {
|
|
const sb = createClient(process.env.SUPABASE_URL, process.env.SUPABASE_SERVICE_KEY,
|
|
{ auth: { persistSession: false } });
|
|
const today = process.env.FIT_AS_OF || new Date().toISOString().slice(0, 10);
|
|
|
|
const rows = await src.loadRows(sb, { sport: SPORT, stat: STAT, modelVersion: MODEL_VERSION, before: today });
|
|
rows.sort((a, b) => (a.date < b.date ? -1 : a.date > b.date ? 1 : (a.id < b.id ? -1 : 1)));
|
|
const audit = src.eraAudit(rows, MODEL_VERSION);
|
|
const dates = [...new Set(rows.map((r) => r.date))].sort();
|
|
|
|
// ── STEP 15 — SAMPLE SUFFICIENCY ───────────────────────────────────────
|
|
const bandN = {};
|
|
for (const [lo, hi] of [[0.50, 0.60], [0.60, 0.70], [0.70, 0.80]]) {
|
|
bandN[`${lo.toFixed(2)}-${hi.toFixed(2)}`] = rows.filter((r) => r.p >= lo && r.p < hi).length;
|
|
}
|
|
console.log(JSON.stringify({ section: 'SOURCE', model_version: MODEL_VERSION, fit_as_of: today,
|
|
rows: rows.length, dates: dates.length, first: dates[0], last: dates[dates.length - 1],
|
|
era_audit: audit, in_support: rows.filter((r) => inSupport(r.p)).length,
|
|
band_n: bandN, min_fit_rows: MIN_FIT }, null, 1));
|
|
|
|
// ── STEPS 14/16/17 — WALK-FORWARD, current era only ────────────────────
|
|
const folds = [];
|
|
const mapAt = Object.fromEntries(PROBES.map((p) => [p, []]));
|
|
for (let f = FOLDS; f >= 1; f--) {
|
|
const endIdx = dates.length - (f - 1) * FOLD_DATES;
|
|
const startIdx = endIdx - FOLD_DATES;
|
|
if (startIdx <= 0) continue;
|
|
const evalDates = dates.slice(startIdx, endIdx);
|
|
const trainRows = rows.filter((r) => r.date < evalDates[0]); // STRICTLY before
|
|
const evalRows = rows.filter((r) => evalDates.includes(r.date) && inSupport(r.p));
|
|
if (trainRows.length < MIN_FIT || evalRows.length < 30) {
|
|
folds.push({ fold: FOLDS - f + 1, eval_dates: evalDates, train_n: trainRows.length,
|
|
eval_n: evalRows.length, refused: 'insufficient' });
|
|
continue;
|
|
}
|
|
const map = cal.fitIsotonic(trainRows, { minTotal: MIN_FIT });
|
|
if (!map) { folds.push({ fold: FOLDS - f + 1, refused: 'fitter refused' }); continue; }
|
|
const ys = evalRows.map((r) => r.won);
|
|
const rawP = evalRows.map((r) => r.p);
|
|
const srv = evalRows.map((r) => cal.applyIsotonic(map, r.p) ?? r.p);
|
|
const ci = pairedCI(srv, rawP, ys);
|
|
// leak check: no training row may be dated at or after the eval block
|
|
const leak = trainRows.filter((r) => r.date >= evalDates[0]).length;
|
|
for (const p of PROBES) { const v = cal.applyIsotonic(map, p); if (v != null) mapAt[p].push(v); }
|
|
const bands = [[0.50, 0.60], [0.60, 0.70], [0.70, 0.80]].map(([lo, hi]) => {
|
|
const sel = evalRows.map((r, i) => (r.p >= lo && r.p < hi ? i : -1)).filter((i) => i >= 0);
|
|
const k = sel.reduce((s, i) => s + ys[i], 0);
|
|
const w = wilson(k, sel.length);
|
|
return { band: `${lo.toFixed(2)}-${hi.toFixed(2)}`, n: sel.length,
|
|
mean_served: sel.length ? r3(sel.reduce((s, i) => s + srv[i], 0) / sel.length) : null,
|
|
observed: sel.length ? r3(k / sel.length) : null,
|
|
observed_ci95: w ? [r3(w[0]), r3(w[1])] : null,
|
|
error: sel.length ? r3(sel.reduce((s, i) => s + srv[i], 0) / sel.length - k / sel.length) : null };
|
|
});
|
|
folds.push({ fold: FOLDS - f + 1, eval_dates: evalDates, train_n: trainRows.length,
|
|
train_through: trainRows[trainRows.length - 1].date, eval_n: evalRows.length,
|
|
future_rows_in_train: leak,
|
|
brier_raw: r5(brier(rawP, ys)), brier_candidate: r5(brier(srv, ys)),
|
|
logloss_raw: r5(logloss(rawP, ys)), logloss_candidate: r5(logloss(srv, ys)),
|
|
ece_raw: r5(ece(rawP, ys)), ece_candidate: r5(ece(srv, ys)),
|
|
delta_brier: r5(brier(srv, ys) - brier(rawP, ys)), ci95: [r5(ci[0]), r5(ci[1])],
|
|
bands });
|
|
}
|
|
console.log(JSON.stringify({ section: 'WALK_FORWARD', folds }, null, 1));
|
|
console.log(JSON.stringify({ section: 'MAPPING_STABILITY',
|
|
probes: Object.fromEntries(PROBES.map((p) => [p, summ(mapAt[p])])) }, null, 1));
|
|
|
|
// ── POOLED across folds, and POLICY A vs POLICY B on the same folds ────
|
|
const pooled = { raw: [], cand: [], y: [] };
|
|
for (let f = FOLDS; f >= 1; f--) {
|
|
const endIdx = dates.length - (f - 1) * FOLD_DATES;
|
|
const startIdx = endIdx - FOLD_DATES;
|
|
if (startIdx <= 0) continue;
|
|
const evalDates = dates.slice(startIdx, endIdx);
|
|
const trainAll = rows.filter((r) => r.date < evalDates[0]);
|
|
const evalRows = rows.filter((r) => evalDates.includes(r.date) && inSupport(r.p));
|
|
if (trainAll.length < MIN_FIT || !evalRows.length) continue;
|
|
const mB = cal.fitIsotonic(trainAll, { minTotal: MIN_FIT }); // POLICY B
|
|
const cutA = Math.floor(trainAll.length * 0.65);
|
|
const mA = cal.fitIsotonic(trainAll.slice(0, cutA), { minTotal: MIN_FIT }); // POLICY A
|
|
if (!mA || !mB) continue;
|
|
for (const r of evalRows) {
|
|
pooled.y.push(r.won); pooled.raw.push(r.p);
|
|
pooled.cand.push({ A: cal.applyIsotonic(mA, r.p) ?? r.p, B: cal.applyIsotonic(mB, r.p) ?? r.p });
|
|
}
|
|
}
|
|
const yy = pooled.y;
|
|
const A = pooled.cand.map((c) => c.A); const B = pooled.cand.map((c) => c.B);
|
|
const ciA = pairedCI(A, pooled.raw, yy); const ciB = pairedCI(B, pooled.raw, yy);
|
|
const ciAB = pairedCI(A, B, yy);
|
|
console.log(JSON.stringify({ section: 'POLICY_A_VS_B_POOLED_FOLDS', n: yy.length,
|
|
brier_raw: r5(brier(pooled.raw, yy)),
|
|
A_era_filtered_65_35: { brier: r5(brier(A, yy)), logloss: r5(logloss(A, yy)), ece: r5(ece(A, yy)),
|
|
delta_vs_raw: r5(brier(A, yy) - brier(pooled.raw, yy)), ci95: [r5(ciA[0]), r5(ciA[1])] },
|
|
B_all_current_era: { brier: r5(brier(B, yy)), logloss: r5(logloss(B, yy)), ece: r5(ece(B, yy)),
|
|
delta_vs_raw: r5(brier(B, yy) - brier(pooled.raw, yy)), ci95: [r5(ciB[0]), r5(ciB[1])] },
|
|
A_minus_B: { delta: r5(brier(A, yy) - brier(B, yy)), ci95: [r5(ciAB[0]), r5(ciAB[1])] },
|
|
}, null, 1));
|
|
process.exit(0);
|
|
})().catch((e) => { console.error(e); process.exit(1); });
|