Files
vyndr/scripts/estimator-adjudication.js
T
builtbykev a8de676756 A probability is served because evidence supports it, not because nothing else answered
The band gate was blocked for its `else` branch. It read:

    candidate = F(raw)
    served    = inCertifiedBand(candidate) ? candidate : RAW

and above raw 0.60 the model is measured overconfident — holdout raw 0.80-0.90
predicts 0.843 and realizes 0.639. So "the calibrator is not supported here" was
being answered with a number already proven wrong. Unsupported calibration does
not make raw true.

Four candidates were adjudicated on ONE split — fit on the earliest 60% of
train, decide support on the last 40%, evaluate on a holdout that saw neither:

  A low-param      80.2% coverage  0.24374  REFUTED — its extra region
                   (raw 0.80-0.90) certified on cert (err +0.040, n=55) and
                   refuted on holdout (served 0.754 vs observed 0.639), and it
                   leaves a hole at 0.70-0.80 while serving the island above it
  B isotonic       91.3% coverage  0.24337  CERTIFIED, contiguous raw [0.50,0.80)
  C empirical band 91.3% coverage  0.24335  REFUTED — refitted point-in-time on
                   current-model hits the realized rates INVERT in grade order
                   (B+ 0.593 < B 0.614 < C+ 0.623), so the served function steps
                   down at raw 0.78. Its shipped constants come from 3,417 props
                   pooled across four batter stats and do not reproduce here
  D raw identity   43.1% coverage  0.24866  certifies raw 0.50-0.60 and only there

Raw is candidate D, not a fallback. It earns exactly one region (holdout error
+0.010 on n=1,316), which is why the law is "raw must earn its region" rather
than "raw is never true". B already covers that region, so no hybrid is built.

Above raw 0.80 nothing is certified and nothing is served. That is the region
where raw is most wrong, isotonic over-corrects (cert err -0.093) and its LODO
mapping at 0.95 has spread 0.180. 8.7% of holdout rows land there.

The registry did not need changing. `serves(stat, p)` already tested certified
bands against the RAW p_win — support in the input domain, the correct question —
and returned {serve:false, reason}. It never said "serve raw". The output-space
gate and the raw fallback were both invented downstream in calibrationService.

ACTIVATION IS OFF. PROBABILITY_CONTRACT_SHADOW defaults to 0, CALIBRATION_DEPLOYED
stays frozen empty, and every served field is byte-identical. This releases the
support first, which is the required order. The shadow records raw belief, the
candidate served value, the state, the estimator identity, and what EV/Kelly/VALUE
would be under the actionability law — into its own column, read by nothing.

Migration 051 was applied to production BEFORE retentionService named the column.
PostgREST builds a bulk insert from the first row's shape, so a key whose column
does not exist 400s the whole batch silently — that is how migration 038 took
retention down for three days.

The user-facing contradiction is NOT fixed here. A B+ still says "realized about
66%" beside a confidence of 84. Fixing that is activation, and activation costs
32% of VALUE flags and 46% of Kelly recommendations on the holdout.

Suite 401/401, 5,580 passed, 4 skipped, deterministic across three runs.
Teeth 23/23, each independently injected and restored byte-identically.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CQJeAG8vcDoL5zkiaJyVb8
2026-09-02 21:03:16 -04:00

299 lines
17 KiB
JavaScript

#!/usr/bin/env node
'use strict';
/**
* estimator-adjudication — FOUR CANDIDATES, ONE SPLIT, NO RAW FALLBACK.
*
* The band gate is blocked because its `else` branch serves RAW, and raw is
* measured overconfident above ~0.60. So here RAW is not a fallback: it is
* CANDIDATE D, and it must earn each region on evidence like any other.
* Where nothing certifies, the contract returns NO_SERVED_PROBABILITY.
*
* ONE SPLIT FOR EVERY CANDIDATE (Step 6):
* FIT earliest 60% of TRAIN -> derive the estimator
* CERT latest 40% of TRAIN -> decide which raw regions are supported
* HOLD >= SPLIT -> evaluate the resulting contract, untouched
*
* SUPPORT IS IN THE RAW INPUT DOMAIN (Step 8). An estimator may not certify
* itself by emitting a number that happens to land in a preferred interval.
*
* RUN FROM THE DEPLOYED RELEASE WORKTREE so the fitters, grade bands and EV
* functions exercised are the ones production runs.
*/
require('dotenv').config({ quiet: true });
const { createClient } = require('@supabase/supabase-js');
const cal = require('../src/services/model/calibration');
const lpc = require('../src/services/model/lowParamCalibrator');
const sg = require('../src/services/model/servedGrade');
const { evPct } = require('../src/utils/devig');
const { isValue, isTakeable } = require('../src/config/valueEngine');
const { quarterKelly } = require('../src/utils/kelly');
const { paginate } = require('../src/utils/safePaginate');
const { uniqueKeyFor } = require('../src/utils/tableKeys');
const SB_URL = process.env.SUPABASE_URL;
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
const ERA = process.env.CAL_ERA || 'engine1@2026-08-07-fullwindow';
const ERA_START = process.env.CAL_ERA_START || '2026-08-11';
const SPLIT = process.env.CAL_SPLIT || '2026-08-22';
const TOL = Number(process.env.CAL_TOL || 0.05); // certifyBands tolerance
const MIN_BIN = Number(process.env.CAL_MIN_BIN || 40); // certifyBands minBin
const PAGE = 1000;
const r3 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 1000) / 1000);
const r5 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 100000) / 100000);
async function pageSafe(sb, apply) {
return paginate(() => apply(sb.from('ledger_entries')
.select('id, p_win, outcome, game_date, side, line, locked_odds, model_version, grade, quarantine_reason')),
{ key: uniqueKeyFor('ledger_entries'), pageSize: PAGE, label: 'estimator-adjudication' });
}
const READS = {
ledger: (sb) => pageSafe(sb, (q) => q.eq('sport', 'mlb').is('user_id', null).eq('stat', 'hits')
.in('outcome', ['hit', 'miss']).not('p_win', 'is', null)
.eq('model_version', ERA).gte('game_date', ERA_START)),
};
// ── metrics ───────────────────────────────────────────────────────────────
const brier = (ps, ys) => (ps.length ? ps.reduce((s, p, i) => s + (p - ys[i]) ** 2, 0) / ps.length : null);
function logloss(ps, ys) { const E = 1e-12; let s = 0;
for (let i = 0; i < ps.length; i++) { const p = Math.min(1 - E, Math.max(E, ps[i])); s += -(ys[i] * Math.log(p) + (1 - ys[i]) * Math.log(1 - p)); }
return ps.length ? s / ps.length : null; }
function ece(ps, ys, bins = 10) { const a = Array.from({ length: bins }, () => ({ n: 0, sp: 0, sy: 0 }));
for (let i = 0; i < ps.length; i++) { const b = Math.min(bins - 1, Math.floor(ps[i] * bins)); a[b].n++; a[b].sp += ps[i]; a[b].sy += ys[i]; }
let e = 0; for (const b of a) if (b.n) e += (b.n / ps.length) * Math.abs(b.sp / b.n - b.sy / b.n); return e; }
function wilson(k, n, z = 1.96) { if (!n) return null; const p = k / n, d = 1 + z * z / n;
const c = (p + z * z / (2 * n)) / d, h = (z * Math.sqrt(p * (1 - p) / n + z * z / (4 * n * n))) / d;
return [Math.max(0, c - h), Math.min(1, c + h)]; }
function pairedCI(a, b, ys, iters = 2000, seed = 7) { let s = seed >>> 0;
const rnd = () => { s = (s * 1664525 + 1013904223) >>> 0; return s / 4294967296; };
const n = ys.length, out = [];
for (let it = 0; it < iters; it++) { let sa = 0, sb = 0;
for (let i = 0; i < n; i++) { const j = Math.floor(rnd() * n); sa += (a[j] - ys[j]) ** 2; sb += (b[j] - ys[j]) ** 2; }
out.push(sa / n - sb / n); }
out.sort((x, y) => x - y); return [out[Math.floor(iters * 0.025)], out[Math.floor(iters * 0.975)]]; }
// ── RAW-DOMAIN SUPPORT (Step 8) ───────────────────────────────────────────
// Bin CERT rows by their RAW value; a bin is supported when it has enough
// evidence AND the estimator's output there matches what actually happened.
const RAW_BINS = [[0.00, 0.50], [0.50, 0.60], [0.60, 0.70], [0.70, 0.80], [0.80, 0.90], [0.90, 1.01]];
function certifySupport(certRows, apply) {
return RAW_BINS.map(([lo, hi]) => {
const rows = certRows.filter((r) => r.p >= lo && r.p < hi);
const vals = rows.map((r) => apply(r.p)).filter((v) => v != null);
if (vals.length !== rows.length || rows.length === 0) {
return { lo, hi, n: rows.length, supported: false, reason: rows.length ? 'estimator_undefined' : 'no_evidence' };
}
const meanP = vals.reduce((s, v) => s + v, 0) / vals.length;
const obs = rows.reduce((s, r) => s + r.won, 0) / rows.length;
const err = meanP - obs;
const ci = wilson(rows.reduce((s, r) => s + r.won, 0), rows.length);
const enough = rows.length >= MIN_BIN;
const close = Math.abs(err) <= TOL;
return { lo, hi, n: rows.length, mean_estimate: r3(meanP), observed: r3(obs), error: r3(err),
observed_ci95: ci ? [r3(ci[0]), r3(ci[1])] : null,
supported: enough && close, reason: !enough ? 'insufficient_evidence' : (!close ? 'error_exceeds_tolerance' : null) };
});
}
const supportedAt = (support, p) => support.some((b) => b.supported && p >= b.lo && p < b.hi);
/** The contract: certified region -> number; everywhere else -> UNAVAILABLE. */
function makeContract(apply, support, state) {
return (p) => {
if (!supportedAt(support, p)) return { served: null, state: 'NO_SERVED_PROBABILITY' };
const v = apply(p);
if (v == null) return { served: null, state: 'UNSUPPORTED' };
return { served: v, state };
};
}
/** Non-decreasing across supported points; a GAP is not a backward move. */
function monotonicity(C, lo = 0.30, hi = 1.0, step = 0.001) {
let prev = null, prevAt = null, maxDrop = 0, at = null, maxStep = 0, stepAt = null, covered = 0, total = 0;
for (let x = lo; x <= hi + 1e-9; x += step) {
const p = Math.round(x * 1000) / 1000; total++;
const v = C(p);
if (v.served == null) continue;
covered++;
if (prev != null) { const d = v.served - prev;
if (d < maxDrop) { maxDrop = d; at = p; }
if (Math.abs(d) > Math.abs(maxStep)) { maxStep = d; stepAt = p; } }
prev = v.served; prevAt = p;
}
return { monotone: maxDrop >= -1e-9, max_downward: r5(maxDrop), max_downward_at: at,
max_step: r5(maxStep), max_step_at: stepAt, grid_covered_pct: r3(covered / total) };
}
// ── CANDIDATE C — grade-band histogram, RECOMPUTED point-in-time ──────────
const GRADE_MIN = sg.BANDS.map((b) => ({ letter: b.letter, min: b.min, shipped_realized: b.realized }));
function letterFor(p) { const b = sg.BANDS.find((x) => p >= x.min) || sg.BANDS[sg.BANDS.length - 1]; return b.letter; }
function fitGradeBand(fitRows) {
const acc = new Map();
for (const r of fitRows) { const L = letterFor(r.p);
if (!acc.has(L)) acc.set(L, { n: 0, k: 0 }); const a = acc.get(L); a.n++; a.k += r.won; }
const table = GRADE_MIN.map((g) => { const a = acc.get(g.letter) || { n: 0, k: 0 };
const rate = a.n ? a.k / a.n : null; const ci = wilson(a.k, a.n);
return { letter: g.letter, min: g.min, n: a.n, fitted_realized: r3(rate),
ci95: ci ? [r3(ci[0]), r3(ci[1])] : null, shipped_realized: g.shipped_realized }; });
return table;
}
function gradeBandApply(table) {
return (p) => { const g = table.find((t) => p >= t.min) || table[table.length - 1];
return (g && g.n >= MIN_BIN && g.fitted_realized != null) ? g.fitted_realized : null; };
}
async function main() {
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
const raw = await READS.ledger(sb);
const all = raw.filter((r) => r.quarantine_reason == null)
.map((r) => ({ p: Number(r.p_win), won: r.outcome === 'hit' ? 1 : 0,
d: String(r.game_date), date: String(r.game_date), side: r.side, grade: r.grade,
odds: r.locked_odds == null ? null : Number(r.locked_odds) }))
.filter((r) => Number.isFinite(r.p))
.sort((a, b) => a.d.localeCompare(b.d));
const train = all.filter((r) => r.d < SPLIT);
const hold = all.filter((r) => r.d >= SPLIT);
const cut = Math.floor(train.length * 0.60);
const fit = train.slice(0, cut), cert = train.slice(cut);
console.log(JSON.stringify({ section: 'SPLIT', era: ERA, n: all.length,
dates: [...new Set(all.map((r) => r.d))].length,
fit_n: fit.length, fit_through: fit[fit.length - 1].d,
cert_n: cert.length, cert_from: cert[0].d, cert_through: cert[cert.length - 1].d,
hold_n: hold.length, hold_from: hold[0].d, hold_through: hold[hold.length - 1].d,
hold_dates: [...new Set(hold.map((r) => r.d))].length,
tolerance: TOL, min_bin: MIN_BIN }, null, 1));
// derive every candidate on the SAME fit window
const isoMap = cal.fitIsotonic(fit, { minTotal: 200 });
const lpModel = lpc.fitPlatt(fit, {});
const gbTable = fitGradeBand(fit);
const applyIso = (p) => cal.applyIsotonic(isoMap, p);
const applyLp = (p) => lpc.applyPlatt(lpModel, p);
const applyGb = gradeBandApply(gbTable);
const applyRaw = (p) => p;
console.log(JSON.stringify({ section: 'CANDIDATE_C_GRADE_BAND_REFIT',
provenance_of_shipped_constants: '3,417 settled props POOLED ACROSS FOUR BATTER STATS, static, not hits-specific and not current-model specific',
refit_on_fit_window: gbTable,
monotone_in_grade_order: (() => { const v = gbTable.filter((t) => t.fitted_realized != null).map((t) => t.fitted_realized);
let ok = true; for (let i = 1; i < v.length; i++) if (v[i] > v[i - 1]) ok = false; return ok; })(),
}, null, 1));
const candidates = [
{ id: 'A_LOW_PARAM', state: 'CERTIFIED_CALIBRATED', apply: applyLp },
{ id: 'B_ISOTONIC', state: 'CERTIFIED_CALIBRATED', apply: applyIso },
{ id: 'C_EMPIRICAL_BAND', state: 'CERTIFIED_EMPIRICAL_BAND', apply: applyGb },
{ id: 'D_RAW_IDENTITY', state: 'CERTIFIED_RAW', apply: applyRaw },
];
for (const c of candidates) { c.support = certifySupport(cert, c.apply); c.C = makeContract(c.apply, c.support, c.state); }
console.log(JSON.stringify({ section: 'SUPPORT_RAW_DOMAIN',
candidates: Object.fromEntries(candidates.map((c) => [c.id, c.support])) }, null, 1));
// ── HOLDOUT EVALUATION, covered rows only + coverage stated ─────────────
const evalRows = (C) => { const cov = [], ys = [];
for (const r of hold) { const v = C(r.p); if (v.served != null) { cov.push(v.served); ys.push(r.won); } }
return { cov, ys }; };
const out = {};
for (const c of candidates) {
const { cov, ys } = evalRows(c.C);
const rawOnSame = hold.filter((r) => c.C(r.p).served != null).map((r) => r.p);
const e = { coverage_n: cov.length, coverage_pct: r3(cov.length / hold.length),
brier: r5(brier(cov, ys)), logloss: r5(logloss(cov, ys)), ece: r5(ece(cov, ys)),
brier_of_raw_on_same_rows: r5(brier(rawOnSame, ys)) };
if (cov.length && c.id !== 'D_RAW_IDENTITY') {
const ci = pairedCI(cov, rawOnSame, ys);
e.delta_vs_raw_on_covered = r5(brier(cov, ys) - brier(rawOnSame, ys));
e.ci95 = [r5(ci[0]), r5(ci[1])];
}
e.bands = RAW_BINS.map(([lo, hi]) => {
const rows = hold.filter((r) => r.p >= lo && r.p < hi);
const served = rows.map((r) => c.C(r.p)).filter((v) => v.served != null);
const covRows = rows.filter((r) => c.C(r.p).served != null);
const k = covRows.reduce((s, r) => s + r.won, 0);
const ci = wilson(k, covRows.length);
return { band: `${lo.toFixed(2)}-${hi >= 1 ? '1.00' : hi.toFixed(2)}`, holdout_n: rows.length,
covered_n: covRows.length,
mean_served: served.length ? r3(served.reduce((s, v) => s + v.served, 0) / served.length) : null,
observed: covRows.length ? r3(k / covRows.length) : null,
observed_ci95: ci ? [r3(ci[0]), r3(ci[1])] : null,
calibration_error: served.length && covRows.length
? r3(served.reduce((s, v) => s + v.served, 0) / served.length - k / covRows.length) : null };
});
e.monotonicity = monotonicity(c.C);
out[c.id] = e;
}
console.log(JSON.stringify({ section: 'HOLDOUT_EVAL', holdout_n: hold.length, candidates: out }, null, 1));
// ── LODO TAIL (Step 10) ────────────────────────────────────────────────
const probes = [0.70, 0.75, 0.80, 0.85, 0.90, 0.95];
const trainDates = [...new Set(train.map((r) => r.d))].sort();
const lodo = { A_LOW_PARAM: {}, B_ISOTONIC: {}, C_EMPIRICAL_BAND: {}, D_RAW_IDENTITY: {} };
for (const k of Object.keys(lodo)) for (const t of probes) lodo[k][t] = [];
for (const dd of trainDates) {
const sub = train.filter((r) => r.d !== dd);
const f2 = sub.slice(0, Math.floor(sub.length * 0.60));
const m2 = cal.fitIsotonic(f2, { minTotal: 200 });
const l2 = lpc.fitPlatt(f2, {});
const g2 = gradeBandApply(fitGradeBand(f2));
for (const t of probes) {
if (m2) { const v = cal.applyIsotonic(m2, t); if (v != null) lodo.B_ISOTONIC[t].push(v); }
if (l2) { const v = lpc.applyPlatt(l2, t); if (v != null) lodo.A_LOW_PARAM[t].push(v); }
const v3 = g2(t); if (v3 != null) lodo.C_EMPIRICAL_BAND[t].push(v3);
lodo.D_RAW_IDENTITY[t].push(t);
}
}
const summ = (a) => { if (!a.length) return { support: 0 };
const s = [...a].sort((x, y) => x - y); const q = (f) => s[Math.min(s.length - 1, Math.floor(s.length * f))];
return { support: s.length, median: r3(q(0.5)), min: r3(s[0]), max: r3(s[s.length - 1]),
iqr: r3(q(0.75) - q(0.25)), spread: r3(s[s.length - 1] - s[0]) }; };
console.log(JSON.stringify({ section: 'LODO_TAIL', refits: trainDates.length,
candidates: Object.fromEntries(Object.entries(lodo).map(([k, v]) =>
[k, Object.fromEntries(Object.entries(v).map(([t, a]) => [t, summ(a)]))])) }, null, 1));
// ── PRODUCT IMPACT (Steps 24-25) ───────────────────────────────────────
const impact = {};
for (const c of candidates) {
let uncert = 0, evLost = 0, valLost = 0, valGained = 0, kellyLost = 0, gradeChanged = 0, sideChanged = 0;
let evAbs = 0, evBoth = 0;
const uncertByBand = {}, uncertByGrade = {};
for (const r of hold) {
const v = c.C(r.p);
if (v.served == null) {
uncert++;
const b = RAW_BINS.find(([lo, hi]) => r.p >= lo && r.p < hi);
const key = `${b[0].toFixed(2)}-${b[1] >= 1 ? '1.00' : b[1].toFixed(2)}`;
uncertByBand[key] = (uncertByBand[key] || 0) + 1;
const L = letterFor(r.p); uncertByGrade[L] = (uncertByGrade[L] || 0) + 1;
if (r.odds != null) { if (evPct(r.p, r.odds) != null) evLost++;
if (isValue(r.odds, evPct(r.p, r.odds))) valLost++;
if (quarterKelly(r.p, r.odds)) kellyLost++; }
continue;
}
if (letterFor(r.p) !== letterFor(r.p)) gradeChanged++; // grade stays raw-derived by law
if ((r.p > 0.5) !== (v.served > 0.5)) { /* served value crossing 0.5 is not a side change */ }
if (r.odds != null) { const e0 = evPct(r.p, r.odds), e1 = evPct(v.served, r.odds);
if (e0 != null && e1 != null) { evAbs += Math.abs(e1 - e0); evBoth++; }
const w0 = isValue(r.odds, e0), w1 = isValue(r.odds, e1);
if (w0 && !w1) valLost++; if (!w0 && w1) valGained++;
if (quarterKelly(r.p, r.odds) && !quarterKelly(v.served, r.odds)) kellyLost++; }
}
impact[c.id] = { uncertified_rows: uncert, uncertified_pct: r3(uncert / hold.length),
uncertified_by_raw_band: uncertByBand, uncertified_by_grade: uncertByGrade,
ev_withdrawn: evLost, value_withdrawn: valLost, value_gained: valGained,
kelly_withdrawn: kellyLost, mean_abs_ev_change_on_served: r3(evAbs / (evBoth || 1)),
grade_changes: gradeChanged, side_changes: sideChanged };
}
const withOdds = hold.filter((r) => r.odds != null);
console.log(JSON.stringify({ section: 'PRODUCT_IMPACT', holdout_n: hold.length,
baseline: { with_odds: withOdds.length, takeable: withOdds.filter((r) => isTakeable(r.odds)).length,
value_raw: withOdds.filter((r) => isValue(r.odds, evPct(r.p, r.odds))).length,
kelly_raw: withOdds.filter((r) => quarterKelly(r.p, r.odds)).length },
candidates: impact }, null, 1));
process.exit(0);
}
if (require.main === module) main().catch((e) => { console.error(e); process.exit(1); });
module.exports = { READS, certifySupport, makeContract, monotonicity, fitGradeBand, gradeBandApply, letterFor, RAW_BINS };