From 556d186ff1e0599aa6c202f2c89c537767e9cf43 Mon Sep 17 00:00:00 2001 From: Kev Date: Wed, 2 Sep 2026 22:53:21 -0400 Subject: [PATCH] One evaluator for the shadow, so the probe cannot report a mode the pipeline is not in MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit "Is the shadow effective?" was only answerable by waiting for a snapshot to write a row. That leaves a blind spot with real cost: a variable SET IN COOLIFY BUT NOT YET APPLIED to the running process is indistinguishable from an unset one, and the runtime probe already proves the distinction matters — code_sha 22cf51c with started_at 01:45:44Z means anything set after that is not in this process's environment. probabilityContract.shadowState() is now the single evaluator. snapshotService calls it and the protected status probe calls it, and a test asserts NEITHER reads process.env directly — the same rule that keeps lineage_write_mode honest. Reading the env in two places is how a status page and a gate come to disagree. Strict by construction: only the exact string '1' enables it. 'true', 'yes', 'on', '01', ' 1 ' and '' are all OFF, because a loose parse turns a typo into an activation. `configuration_source` separates an unset variable from one explicitly set to '0', and `live_serving` is reported as its own switch so the shadow can never be read as implying serving. No behaviour changes. The shadow still defaults OFF, CALIBRATION_DEPLOYED is still [], and served fields are untouched. Frontend byte-identical to the last green build (git reports zero changes under web/), so the build from 22cf51c stands. Suite 401/401, 5,597 passed, 4 skipped. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01CQJeAG8vcDoL5zkiaJyVb8 --- scripts/fit-policy-adjudication.js | 166 +++++++++++++++++++ scripts/teeth-certified-probability.js | 4 +- src/routes/internal.js | 5 + src/services/model/probabilityContract.js | 31 +++- src/services/snapshotService.js | 2 +- tests/unit/probabilityContractShadow.test.js | 36 +++- 6 files changed, 240 insertions(+), 4 deletions(-) create mode 100644 scripts/fit-policy-adjudication.js diff --git a/scripts/fit-policy-adjudication.js b/scripts/fit-policy-adjudication.js new file mode 100644 index 0000000..961ce19 --- /dev/null +++ b/scripts/fit-policy-adjudication.js @@ -0,0 +1,166 @@ +#!/usr/bin/env node +'use strict'; +/** + * fit-policy-adjudication — MEASURE the production fitting policy, do not change it. + * + * The artifact now has an identity. The open question is whether the PROCEDURE + * that produces tomorrow's curve deserves continuous trust. + * + * Candidates are fitted SIDE-EFFECT-FREE as of the production `fit_as_of`, and + * compared both on their mapping and on point-in-time out-of-sample score. + * No policy is written anywhere by this script. + */ +require('dotenv').config({ quiet: true }); +const { createClient } = require('@supabase/supabase-js'); +const crypto = require('crypto'); +const cal = require('../src/services/model/calibration'); +const calSvc = require('../src/services/model/calibrationService'); + +const ERA = 'engine1@2026-08-07-fullwindow'; +const FIT_AS_OF = process.env.FIT_AS_OF || '2026-09-02'; +const PROBES = [0.50, 0.55, 0.60, 0.65, 0.70, 0.75, 0.79]; +const r3 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 1000) / 1000); +const r5 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 100000) / 100000); +const dg = (o) => crypto.createHash('sha256').update(JSON.stringify(o)).digest('hex').slice(0, 16); + +const brier = (ps, ys) => (ps.length ? ps.reduce((s, p, i) => s + (p - ys[i]) ** 2, 0) / ps.length : null); +function logloss(ps, ys) { const E = 1e-12; let s = 0; + for (let i = 0; i < ps.length; i++) { const p = Math.min(1 - E, Math.max(E, ps[i])); + s += -(ys[i] * Math.log(p) + (1 - ys[i]) * Math.log(1 - p)); } return ps.length ? s / ps.length : null; } +function ece(ps, ys, bins = 10) { const a = Array.from({ length: bins }, () => ({ n: 0, sp: 0, sy: 0 })); + for (let i = 0; i < ps.length; i++) { const b = Math.min(bins - 1, Math.floor(ps[i] * bins)); + a[b].n++; a[b].sp += ps[i]; a[b].sy += ys[i]; } + let e = 0; for (const b of a) if (b.n) e += (b.n / ps.length) * Math.abs(b.sp / b.n - b.sy / b.n); return e; } +function pairedCI(a, b, ys, iters = 2000, seed = 11) { let s = seed >>> 0; + const rnd = () => { s = (s * 1664525 + 1013904223) >>> 0; return s / 4294967296; }; + const n = ys.length, out = []; + for (let it = 0; it < iters; it++) { let sa = 0, sb = 0; + for (let i = 0; i < n; i++) { const j = Math.floor(rnd() * n); sa += (a[j] - ys[j]) ** 2; sb += (b[j] - ys[j]) ** 2; } + out.push(sa / n - sb / n); } + out.sort((x, y) => x - y); return [out[Math.floor(iters * 0.025)], out[Math.floor(iters * 0.975)]]; } + +(async () => { + const sb = createClient(process.env.SUPABASE_URL, process.env.SUPABASE_SERVICE_KEY, + { auth: { persistSession: false } }); + + // The production walk, verbatim — no model filter, strictly before the cutoff. + const raw = await calSvc.loadSettledRows(sb, { sport: 'mlb', stat: 'hits', before: FIT_AS_OF }); + + // ERA COMES FROM THE WALK, NOT FROM AN `.in('id', [...])` BACKFILL. + // A chunked id filter is a URL, and 500 UUIDs is an 18,000-character request + // the fetch layer rejects — chunks return nothing and the rows read as + // "unknown era". Measured here first time round: it reported the superseded + // era as 2.1% of the fit when the true share is over half. + const { paginate } = require('../src/utils/safePaginate'); + const withEra = await paginate( + () => sb.from('ledger_entries') + .select('id, p_win, outcome, game_date, quarantine_reason, model_version') + .eq('sport', 'mlb').is('user_id', null).eq('stat', 'hits') + .in('outcome', ['hit', 'miss']).not('p_win', 'is', null) + .lt('game_date', FIT_AS_OF), + { key: 'id', pageSize: 1000, label: 'fit-policy era walk' }); + if (withEra.length !== raw.length) { + throw new Error(`era walk (${withEra.length}) disagrees with the production walk (${raw.length})`); + } + const all = withEra + .filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book')) + .map((r) => ({ p: Number(r.p_win), won: r.outcome === 'hit' ? 1 : 0, + date: String(r.game_date), era: r.model_version || null })) + .filter((r) => Number.isFinite(r.p)) + .sort((a, b) => (a.date < b.date ? -1 : a.date > b.date ? 1 : 0)); + + const eraCounts = {}; + for (const r of all) eraCounts[r.era || 'unknown'] = (eraCounts[r.era || 'unknown'] || 0) + 1; + + // ── STEP 16 — THE CURRENT POLICY, EXACTLY ────────────────────────────── + const cut = Math.floor(all.length * 0.65); + const curFit = all.slice(0, cut); + const curHeld = all.slice(cut); + const curEraMix = {}; + for (const r of curFit) curEraMix[r.era || 'unknown'] = (curEraMix[r.era || 'unknown'] || 0) + 1; + + console.log(JSON.stringify({ section: 'CURRENT_POLICY', fit_as_of: FIT_AS_OF, + query: { table: 'ledger_entries', filters: ["sport=mlb", "user_id IS NULL", "stat=hits", + "outcome IN (hit,miss)", "p_win NOT NULL", `game_date < ${FIT_AS_OF}`], + model_version_filter: 'NONE — the fit pools every model era' }, + rows_walked: raw.length, rows_clean: all.length, era_counts: eraCounts, + fit_fraction: 0.65, fit_n: curFit.length, withheld_n: curHeld.length, + fit_first_date: curFit[0].date, fit_last_date: curFit[curFit.length - 1].date, + withheld_first_date: curHeld[0].date, withheld_last_date: curHeld[curHeld.length - 1].date, + fit_era_mix: curEraMix, + superseded_era_share_of_fit: r3((curEraMix['engine1@2026-07-20'] || 0) / curFit.length), + refit_cadence: 'once per snapshot run (probabilityContractService.build), cutoff = todayEt()', + object: 'A — a continuously refitted estimator PROCEDURE, not a frozen artifact', + }, null, 1)); + + // ── STEPS 18/19 — CANDIDATE POLICIES, side-effect-free ──────────────── + const eraRows = all.filter((r) => r.era === ERA); + const policies = []; + const mk = (id, desc, fitRows, heldRows) => { + const map = cal.fitIsotonic(fitRows, { minTotal: 200 }); + return { id, desc, fit_n: fitRows.length, held_n: heldRows.length, + fit_last_date: fitRows.length ? fitRows[fitRows.length - 1].date : null, + map, knot_count: map ? map.length : null, knot_digest: map ? dg(map) : null, + probes: Object.fromEntries(PROBES.map((p) => [p, map ? r3(cal.applyIsotonic(map, p)) : null])) }; + }; + policies.push(mk('A_CURRENT_65_35_ALL_ERAS', 'production today: 65% of ALL eras pooled', curFit, curHeld)); + const eCut = Math.floor(eraRows.length * 0.65); + policies.push(mk('B_ERA_FILTERED_65_35', 'same split, current model era only', eraRows.slice(0, eCut), eraRows.slice(eCut))); + // C: expanding train, fixed RECENT holdout (era-filtered) — holdout by DATE, not fraction + const eraDates = [...new Set(eraRows.map((r) => r.date))].sort(); + const holdFrom = eraDates[Math.max(0, eraDates.length - 4)]; // last 4 dates held + policies.push(mk('C_ERA_EXPANDING_FIXED_RECENT_HOLDOUT', 'current era, all but the last 4 settled dates', + eraRows.filter((r) => r.date < holdFrom), eraRows.filter((r) => r.date >= holdFrom))); + // D: era-filtered, ALL rows fitted (no withhold at all) + policies.push(mk('D_ERA_ALL_ROWS_NO_WITHHOLD', 'current era, every settled row fitted', eraRows, [])); + + console.log(JSON.stringify({ section: 'CANDIDATE_MAPS', era_rows: eraRows.length, + holdout_from_date_for_C: holdFrom, + policies: policies.map(({ map, ...rest }) => rest) }, null, 1)); + + const base = policies[0]; + console.log(JSON.stringify({ section: 'MAP_DELTAS_VS_PRODUCTION', + deltas: policies.slice(1).map((p) => ({ id: p.id, + per_probe: Object.fromEntries(PROBES.map((x) => [x, + (p.probes[x] == null || base.probes[x] == null) ? null : r3(p.probes[x] - base.probes[x])])), + max_abs: r3(Math.max(...PROBES.map((x) => Math.abs((p.probes[x] ?? 0) - (base.probes[x] ?? 0))))) })) }, null, 1)); + + // ── OUT-OF-SAMPLE, POINT IN TIME ─────────────────────────────────────── + // Every policy is fitted on evidence strictly before EVAL_FROM and scored on + // current-era rows at/after it. Same rows for every policy. + const EVAL_FROM = eraDates[Math.max(0, eraDates.length - 4)]; + const evalRows = eraRows.filter((r) => r.date >= EVAL_FROM); + const trainAll = all.filter((r) => r.date < EVAL_FROM); + const trainEra = eraRows.filter((r) => r.date < EVAL_FROM); + const oos = []; + const fitFor = (id) => { + if (id === 'A_CURRENT_65_35_ALL_ERAS') return trainAll.slice(0, Math.floor(trainAll.length * 0.65)); + if (id === 'B_ERA_FILTERED_65_35') return trainEra.slice(0, Math.floor(trainEra.length * 0.65)); + if (id === 'C_ERA_EXPANDING_FIXED_RECENT_HOLDOUT') return trainEra; + return trainEra; + }; + const ys = evalRows.map((r) => r.won); + const rawPs = evalRows.map((r) => r.p); + const inSupport = (p) => p >= 0.50 && p < 0.80; + const covIdx = evalRows.map((r, i) => (inSupport(r.p) ? i : -1)).filter((i) => i >= 0); + const covY = covIdx.map((i) => ys[i]); + const covRaw = covIdx.map((i) => rawPs[i]); + for (const id of ['A_CURRENT_65_35_ALL_ERAS', 'B_ERA_FILTERED_65_35', + 'C_ERA_EXPANDING_FIXED_RECENT_HOLDOUT', 'D_ERA_ALL_ROWS_NO_WITHHOLD']) { + const fr = fitFor(id); + const m = cal.fitIsotonic(fr, { minTotal: 200 }); + if (!m) { oos.push({ id, refused: true, fit_n: fr.length }); continue; } + const served = covIdx.map((i) => cal.applyIsotonic(m, rawPs[i]) ?? rawPs[i]); + const ci = pairedCI(served, covRaw, covY); + oos.push({ id, fit_n: fr.length, fit_last_date: fr[fr.length - 1].date, + knot_digest: dg(m), + eval_n: covY.length, eval_from: EVAL_FROM, + brier: r5(brier(served, covY)), logloss: r5(logloss(served, covY)), ece: r5(ece(served, covY)), + delta_vs_raw: r5(brier(served, covY) - brier(covRaw, covY)), ci95: [r5(ci[0]), r5(ci[1])], + probes: Object.fromEntries(PROBES.map((p) => [p, r3(cal.applyIsotonic(m, p))])) }); + } + console.log(JSON.stringify({ section: 'OUT_OF_SAMPLE_POINT_IN_TIME', + eval_from: EVAL_FROM, eval_rows_in_support: covY.length, + brier_raw_on_same_rows: r5(brier(covRaw, covY)), policies: oos }, null, 1)); + process.exit(0); +})().catch((e) => { console.error(e); process.exit(1); }); diff --git a/scripts/teeth-certified-probability.js b/scripts/teeth-certified-probability.js index 11ad6e2..8d47ba2 100644 --- a/scripts/teeth-certified-probability.js +++ b/scripts/teeth-certified-probability.js @@ -119,7 +119,9 @@ logicTooth(16, 'live serving activates before the shadow has passed', () => { // ── the shadow flag itself ─────────────────────────────────────────────── logicTooth(25, 'shadow flag parses loosely or defaults ON', () => { const snap = fs.readFileSync(path.join(ROOT, 'src/services/snapshotService.js'), 'utf8'); - const strict = snap.includes("String(process.env.PROBABILITY_CONTRACT_SHADOW || '') === '1'"); + const pcSrc = fs.readFileSync(path.join(ROOT, 'src/services/model/probabilityContract.js'), 'utf8'); + const strict = pcSrc.includes("String(raw || '') === '1'") + && snap.includes("probabilityContract').shadowState().shadow === 'ON'"); const scoped = snap.includes("&& sp === 'mlb'"); const statScoped = snap.includes("{ sport: 'mlb', stat: 'hits' }"); return { caught: strict && scoped && statScoped, diff --git a/src/routes/internal.js b/src/routes/internal.js index b4a840c..bf72f92 100644 --- a/src/routes/internal.js +++ b/src/routes/internal.js @@ -276,6 +276,11 @@ router.get('/snapshot/status', async (req, res) => { // it from a second parse is how a status surface and a gate come to // disagree, so there is only one. lineage_write_mode: require('../services/lineageWriteMode').state(), + // THE PROBABILITY-CONTRACT SHADOW, from the SAME evaluator the pipeline + // calls. Without it "is the shadow effective?" is only answerable by + // waiting for a snapshot to write, and a set-but-not-restarted variable + // is indistinguishable from an unset one. + probability_contract: require('../services/model/probabilityContract').shadowState(), // INDEPENDENT COVERAGE. Derived from durable retained state, never from // the writer's own counters — an observer that reads the failing writer's // return value cannot see that writer fail. diff --git a/src/services/model/probabilityContract.js b/src/services/model/probabilityContract.js index 81b3290..ad8334b 100644 --- a/src/services/model/probabilityContract.js +++ b/src/services/model/probabilityContract.js @@ -255,7 +255,36 @@ function confidenceDisplay(res) { }; } +/** + * THE ONE EVALUATOR for the shadow's effective state. + * + * `snapshotService` and the status probe both call this, so the surface cannot + * report a mode the pipeline is not in — the same rule that keeps + * `lineage_write_mode` honest. Reading the env in two places is how a status + * page and a gate come to disagree. + * + * Strict: only the exact string '1' enables it. Anything else is OFF, including + * 'true', 'yes' and ' 1 ' — a loose parse turns a typo into an activation. + */ +const SHADOW_ENV = 'PROBABILITY_CONTRACT_SHADOW'; +function shadowState(env = process.env) { + const raw = env[SHADOW_ENV]; + const on = String(raw || '') === '1'; + return Object.freeze({ + shadow: on ? 'ON' : 'OFF', + env_var: SHADOW_ENV, + configuration_source: raw === undefined ? 'default' : 'environment', + scope: on ? Object.freeze({ sport: 'mlb', stat: 'hits' }) : null, + // Serving is a SEPARATE switch and is not implied by the shadow. + live_serving: 'OFF', + certified_support: MLB_HITS.certified_bands, + estimator_version: MLB_HITS.estimator_version, + model_version: MLB_HITS.model_version, + }); +} + module.exports = { - STATE, CERTIFIED_STATES, ESTIMATOR, CONTRACTS, MLB_HITS, + STATE, CERTIFIED_STATES, ESTIMATOR, CONTRACTS, MLB_HITS, SHADOW_ENV, contractFor, inCertifiedRawBand, resolve, isCertified, derivedClaims, confidenceDisplay, + shadowState, }; diff --git a/src/services/snapshotService.js b/src/services/snapshotService.js index 18e893c..039c357 100644 --- a/src/services/snapshotService.js +++ b/src/services/snapshotService.js @@ -858,7 +858,7 @@ async function runSnapshot(sport, opts = {}) { // this releases the support with activation off, which is the required order. // A failure here must never cost the snapshot — it is a measurement layer. let probContract = null; - if (String(process.env.PROBABILITY_CONTRACT_SHADOW || '') === '1' && sp === 'mlb') { + if (require('./model/probabilityContract').shadowState().shadow === 'ON' && sp === 'mlb') { try { const pcs = deps.probabilityContractService || require('./model/probabilityContractService'); const sbc = deps.supabase || require('../utils/supabase').getSupabaseServiceClient(); diff --git a/tests/unit/probabilityContractShadow.test.js b/tests/unit/probabilityContractShadow.test.js index 7de84fb..2f9d718 100644 --- a/tests/unit/probabilityContractShadow.test.js +++ b/tests/unit/probabilityContractShadow.test.js @@ -128,7 +128,7 @@ describe('the flag is OFF by default', () => { it('snapshotService reads PROBABILITY_CONTRACT_SHADOW and defaults to off', () => { const src = require('fs').readFileSync( require('path').join(__dirname, '../../src/services/snapshotService.js'), 'utf8'); - expect(src).toContain("process.env.PROBABILITY_CONTRACT_SHADOW || '') === '1'"); + expect(src).toContain("probabilityContract').shadowState().shadow === 'ON'"); // and it must not be able to write a served field const block = src.slice(src.indexOf('PROBABILITY CONTRACT SHADOW'), src.indexOf('let lineageIndex')); for (const served of ['g.p_win =', 'g.confidence =', 'g.grade =', 'g.ev_pct =', 'g.value =']) { @@ -183,3 +183,37 @@ describe('the artifact must agree with the contract, not merely accompany it', ( expect(r.probability_contract.raw_model_probability).toBe(0.65); }); }); + +describe('one evaluator for the shadow mode', () => { + it('only the exact string 1 enables it — a loose parse turns a typo into an activation', () => { + expect(pc.shadowState({ PROBABILITY_CONTRACT_SHADOW: '1' }).shadow).toBe('ON'); + // process.env values are always strings, so only string inputs are asserted; + // claiming a numeric 1 should be OFF would be asserting a case the parse + // never sees, and one where OFF is not obviously the right answer anyway. + for (const v of ['true', 'yes', 'on', '0', ' 1 ', '', '01', 'TRUE']) { + expect(pc.shadowState({ PROBABILITY_CONTRACT_SHADOW: v }).shadow).toBe('OFF'); + } + expect(pc.shadowState({}).shadow).toBe('OFF'); + }); + + it('distinguishes an unset variable from one explicitly set', () => { + expect(pc.shadowState({}).configuration_source).toBe('default'); + expect(pc.shadowState({ PROBABILITY_CONTRACT_SHADOW: '0' }).configuration_source).toBe('environment'); + }); + + it('the shadow never implies live serving', () => { + expect(pc.shadowState({ PROBABILITY_CONTRACT_SHADOW: '1' }).live_serving).toBe('OFF'); + }); + + it('the pipeline and the status probe call the SAME evaluator, never a second parse', () => { + const fs = require('fs'); const path = require('path'); + const snap = fs.readFileSync(path.join(__dirname, '../../src/services/snapshotService.js'), 'utf8'); + const route = fs.readFileSync(path.join(__dirname, '../../src/routes/internal.js'), 'utf8'); + expect(snap).toContain("probabilityContract').shadowState().shadow === 'ON'"); + expect(route).toContain("probabilityContract').shadowState()"); + // and NEITHER may read the raw env directly + for (const src of [snap, route]) { + expect(src.includes('process.env.PROBABILITY_CONTRACT_SHADOW')).toBe(false); + } + }); +});