f61ec6b391
Seven orders of measurement-first repair. The served grade does not move. A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS: 410-617 of 2,490 duplicated with an equal number never returned, while rows.length matched the server exactly. safePaginate orders on a real unique key, verifies the tuple at runtime, and THROWS on a query error instead of treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn through that reader, and defense_by_direction's distinct-n was likely below the gate floor all along. A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from pg_index (the context tables are dated-composite and had no single unique column). The unordered helper is deleted, not parked. A3 — ledgerService and retentionService defaulted the SAME env var to DIFFERENT versions, so no ledger row ever carried the marker eligibility requires. One source now. model_snapshots settlement moved onto the cron: 15,484 -> 28,894 settled, repaired-champion 0 -> 7,556. A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no row at-or-before the date means the factor does not apply, never the nearest row. Live path unchanged, proven 400/400 on real rows. A5 — factor_inputs freezes what the factor READ, never the multiplier, so an audit can recompute and check. It also recorded the finding: the three hits factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by the resolver and written by nothing. A6/A7 — matchupKeys resolves those keys from the posted lineup plus the schedule's probable pitchers, and fires the factors into a SHADOW freeze: 248 fires on 308 props, 245 of which would move the grade. The served forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test that decides whether they ever go live. Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity harness 34/34. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
399 lines
17 KiB
JavaScript
399 lines
17 KiB
JavaScript
#!/usr/bin/env node
|
||
'use strict';
|
||
|
||
/**
|
||
* champion-ablation — WHERE DOES THE CHAMPION'S RESOLUTION ACTUALLY COME FROM?
|
||
*
|
||
* Convergent evidence says the problem is INPUTS, not shape: hits-v1 refuted,
|
||
* the ladder reliably worse (−0.030), arch-v1 moving 76% of rows to exactly zero
|
||
* effect, contact/environment/opportunity all CI-includes-zero. Every one of
|
||
* those changed the DISTRIBUTION or added a NUDGE. None changed the information.
|
||
* So before building a sixth thing, decompose the champion.
|
||
*
|
||
* THE CHAMPION IS FIVE LINES OF ARITHMETIC (probabilityEstimator):
|
||
*
|
||
* base = empirical frequency of stat > line over the game log
|
||
* weighted = 0.6·base + 0.4·(same frequency over the last 5)
|
||
* p = weighted + oppAdj(±0.03) + homeAdj(±0.015)
|
||
* if cv>0.40: p = 0.9·p + 0.05 (volatile → pull toward 0.50)
|
||
* p_over = clamp(p, 0.10, 0.95); p_win = side==='under' ? 1−p_over : p_over
|
||
*
|
||
* THE ABLATION IS EXACT, NOT A REFIT. Every adjustment is a closed-form function
|
||
* of stored features, and the consistency step is linear, so each layer can be
|
||
* removed analytically from the stored p_win:
|
||
*
|
||
* f(x) = 0.9x + 0.05 ⟹ f(a+b) = f(a) + 0.9b
|
||
*
|
||
* so subtracting an adjustment is subtracting k·adj with k = 0.9 when the
|
||
* consistency pull fired and 1 when it did not. Nothing is re-estimated, no
|
||
* model is refit, and no game log is re-fetched — which also means no lookahead
|
||
* is even possible here.
|
||
*
|
||
* WHAT CANNOT BE ABLATED SEPARATELY, STATED PLAINLY: `base` and `recency` are
|
||
* recoverable only as their blend (`weighted`), because the stored feature
|
||
* vector holds AVERAGES (l5_avg/l20_avg), not frequencies-over-the-line. So the
|
||
* base/recency split is reported as ONE block. That is a real limit of this
|
||
* measurement, not an oversight.
|
||
*
|
||
* CLAMPED ROWS ARE EXCLUDED from the ablation: at p_over ∈ {0.10, 0.95} the
|
||
* inversion is ambiguous, and guessing the pre-clamp value would be fabrication.
|
||
* Their count is reported.
|
||
*
|
||
* SECOND MEASUREMENT — THE MISSING-FEATURE TEST. featureCache computes and
|
||
* RETAINS far more than the champion reads (park_*, weather_*, rest_days,
|
||
* opportunity_drift, ab_per_game, l5/l10/l20 avgs). If any of those correlates
|
||
* with the champion's RESIDUAL (won − p_win), that is signal sitting unused on
|
||
* disk — a MISSING FEATURE. If none do, that is evidence for AT CEILING with
|
||
* respect to everything we currently compute.
|
||
*
|
||
* SUPABASE_URL=... node scripts/champion-ablation.js
|
||
*/
|
||
|
||
require('dotenv').config();
|
||
const { createClient } = require('@supabase/supabase-js');
|
||
const { paginate } = require('../src/utils/safePaginate');
|
||
const { uniqueKeyFor } = require('../src/utils/tableKeys');
|
||
|
||
const SB_URL = process.env.SUPABASE_URL;
|
||
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
|
||
const PAGE = 1000;
|
||
|
||
const CV_VOLATILE_THRESHOLD = 0.40;
|
||
const PROB_FLOOR = 0.10;
|
||
const PROB_CEIL = 0.95;
|
||
const clamp = (p) => Math.max(PROB_FLOOR, Math.min(PROB_CEIL, p));
|
||
|
||
function corr(xs, ys) {
|
||
const n = xs.length;
|
||
if (n < 3) return null;
|
||
const mx = xs.reduce((a, b) => a + b, 0) / n;
|
||
const my = ys.reduce((a, b) => a + b, 0) / n;
|
||
let sxy = 0; let sxx = 0; let syy = 0;
|
||
for (let i = 0; i < n; i += 1) {
|
||
const dx = xs[i] - mx; const dy = ys[i] - my;
|
||
sxy += dx * dy; sxx += dx * dx; syy += dy * dy;
|
||
}
|
||
if (sxx <= 0 || syy <= 0) return null;
|
||
return sxy / Math.sqrt(sxx * syy);
|
||
}
|
||
const r4 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 10000) / 10000);
|
||
const num = (v) => {
|
||
if (v == null || v === '' || typeof v === 'boolean' || typeof v === 'object') return null;
|
||
const n = Number(v);
|
||
return Number.isFinite(n) ? n : null;
|
||
};
|
||
|
||
/** Deterministic xorshift32 — a measurement that changes between runs is not one. */
|
||
function makeRnd(seed) {
|
||
let s = seed >>> 0;
|
||
return () => { s ^= s << 13; s >>>= 0; s ^= s >>> 17; s ^= s << 5; s >>>= 0; return s / 4294967296; };
|
||
}
|
||
|
||
/** Paired bootstrap on a DIFFERENCE of resolutions (same rows → same resample). */
|
||
function bootstrapDiff(rows, keyA, keyB, iters = 3000, seed = 20260803) {
|
||
if (rows.length < 30) return null;
|
||
const rnd = makeRnd(seed);
|
||
const n = rows.length;
|
||
const diffs = [];
|
||
for (let it = 0; it < iters; it += 1) {
|
||
const ys = []; const a = []; const b = [];
|
||
for (let i = 0; i < n; i += 1) {
|
||
const r = rows[Math.floor(rnd() * n)];
|
||
ys.push(r.won); a.push(r[keyA]); b.push(r[keyB]);
|
||
}
|
||
const ca = corr(a, ys); const cb = corr(b, ys);
|
||
if (ca == null || cb == null) continue;
|
||
diffs.push(ca - cb);
|
||
}
|
||
if (diffs.length < 100) return null;
|
||
diffs.sort((x, y) => x - y);
|
||
const q = (p) => r4(diffs[Math.floor(p * (diffs.length - 1))]);
|
||
const ci = [q(0.025), q(0.975)];
|
||
return {
|
||
point: r4(corr(rows.map((r) => r[keyA]), rows.map((r) => r.won))
|
||
- corr(rows.map((r) => r[keyB]), rows.map((r) => r.won))),
|
||
ci95: ci,
|
||
ci_excludes_zero: ci[0] > 0 || ci[1] < 0,
|
||
};
|
||
}
|
||
|
||
/** Bootstrap CI on a single correlation (for the residual-signal test). */
|
||
function bootstrapCorr(rows, key, seed = 20260804, iters = 3000) {
|
||
const usable = rows.filter((r) => r[key] != null);
|
||
if (usable.length < 40) return { n: usable.length, corr: null, ci95: null, ci_excludes_zero: false };
|
||
const rnd = makeRnd(seed);
|
||
const n = usable.length;
|
||
const vals = [];
|
||
for (let it = 0; it < iters; it += 1) {
|
||
const xs = []; const ys = [];
|
||
for (let i = 0; i < n; i += 1) {
|
||
const r = usable[Math.floor(rnd() * n)];
|
||
xs.push(r[key]); ys.push(r.residual);
|
||
}
|
||
const c = corr(xs, ys);
|
||
if (c != null) vals.push(c);
|
||
}
|
||
if (vals.length < 100) return { n, corr: null, ci95: null, ci_excludes_zero: false };
|
||
vals.sort((a, b) => a - b);
|
||
const q = (p) => r4(vals[Math.floor(p * (vals.length - 1))]);
|
||
const ci = [q(0.025), q(0.975)];
|
||
return {
|
||
n,
|
||
corr: r4(corr(usable.map((r) => r[key]), usable.map((r) => r.residual))),
|
||
ci95: ci,
|
||
ci_excludes_zero: ci[0] > 0 || ci[1] < 0,
|
||
};
|
||
}
|
||
|
||
// ── FIX A2 (2026-08-09) — THE SAFE WALK ───────────────────────────────────
|
||
// `page()` above walks with no ORDER BY. Measured on production, that returned
|
||
// the correct row COUNT and the wrong ROWS: up to 33.6% of a read came back
|
||
// twice while an equal share never came back at all. `pageSafe` routes the same
|
||
// call through `src/utils/safePaginate`, which orders on a UNIQUE key on every
|
||
// page, verifies uniqueness at runtime, and THROWS on a query error instead of
|
||
// treating it as end-of-data.
|
||
//
|
||
// `page()` SURVIVES only for the context tables (statcast_aggregates,
|
||
// batter_spray, team_defense, platoon_splits, park_dimensions, game_context...).
|
||
// Those have COMPOSITE primary keys with no single unique column, so
|
||
// safePaginate cannot express them. They measure 0% corruption today; making
|
||
// them safe needs a composite-key ordering the helper does not yet have. Do not
|
||
// use `page()` for ledger_entries or model_snapshots.
|
||
async function pageSafe(sb, table, select, apply, key = uniqueKeyFor(table)) {
|
||
return paginate(() => apply(sb.from(table).select(select)),
|
||
{ key, pageSize: PAGE, label: `${table}` });
|
||
}
|
||
|
||
/** THE MEASURED READS — main() and the harness call the same functions. */
|
||
const READS_LEDGER = {
|
||
ledger: (sb) => pageSafe(sb, 'ledger_entries',
|
||
'id, player_key, stat, line, side, game_date, outcome, quarantine_reason',
|
||
(q) => q.eq('sport', 'mlb').is('user_id', null).in('outcome', ['hit', 'miss'])),
|
||
};
|
||
|
||
/** THE MEASURED READS — main() and the harness call the same functions. */
|
||
const READS = {
|
||
snaps: (sb) => pageSafe(sb, 'model_snapshots',
|
||
'id, player_key, stat, line, side, game_date, p_win, features, quarantine_reason, captured_at, archetype',
|
||
(q) => q.eq('sport', 'mlb').not('p_win', 'is', null).not('features', 'is', null)),
|
||
};
|
||
|
||
const propKey = (r) => `${r.player_key}|${r.stat}|${Number(r.line)}|${String(r.side).toLowerCase()}|${r.game_date}`;
|
||
|
||
/**
|
||
* OUTCOMES COME FROM THE LEDGER, NOT FROM RETENTION.
|
||
*
|
||
* `model_snapshots.outcome` is NULL on all 22,032 rows — the retention table
|
||
* that exists so a different model can be replayed against the same conditions
|
||
* stores the features but was never settled. So the labels are joined from
|
||
* `ledger_entries` on (player_key, stat, line, side, game_date), which is the
|
||
* same identity the ledger's own dedupe constraint uses. Flagged, not fixed —
|
||
* this run is read-only.
|
||
*/
|
||
async function fetchAll(sb) {
|
||
const snaps = await READS.snaps(sb);
|
||
const led = await pageSafe(sb, 'ledger_entries', 'id, player_key, stat, line, side, game_date, outcome, quarantine_reason',
|
||
(q) => q.eq('sport', 'mlb').is('user_id', null).in('outcome', ['hit', 'miss']));
|
||
|
||
const outcomeBy = new Map();
|
||
for (const r of led) {
|
||
if ((r.quarantine_reason || '').startsWith('nontakeable_book')) continue;
|
||
outcomeBy.set(propKey(r), r.outcome);
|
||
}
|
||
return snaps
|
||
.map((r) => ({ ...r, outcome: outcomeBy.get(propKey(r)) || null }))
|
||
.filter((r) => r.outcome === 'hit' || r.outcome === 'miss');
|
||
}
|
||
|
||
function build(rows) {
|
||
// One row per prop — the EARLIEST capture is the lock. Multiple snapshot
|
||
// cycles per day would otherwise weight a prop by how often it was re-graded.
|
||
const byProp = new Map();
|
||
for (const r of rows) {
|
||
if ((r.quarantine_reason || '').startsWith('nontakeable_book')) continue;
|
||
const k = `${r.player_key}|${r.stat}|${r.line}|${r.side}|${r.game_date}`;
|
||
const prev = byProp.get(k);
|
||
if (!prev || String(r.captured_at) < String(prev.captured_at)) byProp.set(k, r);
|
||
}
|
||
|
||
const out = [];
|
||
let clamped = 0;
|
||
for (const r of byProp.values()) {
|
||
const f = r.features || {};
|
||
const pWin = num(r.p_win);
|
||
if (pWin == null) continue;
|
||
const under = String(r.side || '').toLowerCase() === 'under';
|
||
const pOver = under ? 1 - pWin : pWin;
|
||
|
||
// Clamped → the pre-clamp value is unrecoverable. Excluded, counted.
|
||
if (pOver <= PROB_FLOOR + 1e-9 || pOver >= PROB_CEIL - 1e-9) { clamped += 1; continue; }
|
||
|
||
const rank = num(f.opp_rank_stat);
|
||
const oppAdj = rank == null ? 0 : (rank >= 0.70 ? 0.03 : rank <= 0.30 ? -0.03 : 0);
|
||
const ha = num(f.home_away);
|
||
const homeAdj = ha === 1 ? 0.015 : ha === 0 ? -0.015 : 0;
|
||
const sd = num(f.l10_stddev); const l20 = num(f.l20_avg);
|
||
const cv = (sd != null && sd > 0 && l20 != null && l20 > 0) ? sd / l20 : null;
|
||
const consistencyFired = cv != null && cv > CV_VOLATILE_THRESHOLD;
|
||
const k = consistencyFired ? 0.9 : 1;
|
||
|
||
// p_over (unclamped) = f(weighted + oppAdj + homeAdj); f linear ⟹ exact removal.
|
||
const noOpp = pOver - k * oppAdj;
|
||
const noHome = pOver - k * homeAdj;
|
||
const noAdj = pOver - k * oppAdj - k * homeAdj; // = f(weighted)
|
||
// Removing the consistency pull: invert f on the whole thing.
|
||
const noCons = consistencyFired ? (pOver - 0.05) / 0.9 : pOver;
|
||
// base+recency block alone, with every adjustment off.
|
||
const weighted = consistencyFired ? (noAdj - 0.05) / 0.9 : noAdj;
|
||
|
||
const flip = (p) => (under ? 1 - clamp(p) : clamp(p));
|
||
const won = r.outcome === 'hit' ? 1 : 0;
|
||
out.push({
|
||
stat: r.stat,
|
||
won,
|
||
full: pWin,
|
||
no_opp: flip(noOpp),
|
||
no_home: flip(noHome),
|
||
no_consistency: flip(noCons),
|
||
no_adjustments: flip(noAdj),
|
||
weighted_only: flip(weighted),
|
||
residual: won - pWin,
|
||
archetype: r.archetype || null,
|
||
// Features the champion NEVER reads — the missing-feature candidates.
|
||
l5_avg: num(f.l5_avg), l10_avg: num(f.l10_avg), l20_avg: num(f.l20_avg),
|
||
ab_per_game: num(f.ab_per_game), recent_ab_per_game: num(f.recent_ab_per_game),
|
||
opportunity_drift: num(f.opportunity_drift), rest_days: num(f.rest_days),
|
||
game_count_in_7d: num(f.game_count_in_7d),
|
||
park_h: num(f.park_h), park_hr: num(f.park_hr), park_r: num(f.park_r),
|
||
weather_temp_f: num(f.weather_temp_f), weather_wind_mph: num(f.weather_wind_mph),
|
||
weather_precip: num(f.weather_precip),
|
||
// Features it DOES read — controls for the same test.
|
||
opp_rank_stat: num(f.opp_rank_stat), home_away: num(f.home_away),
|
||
l10_stddev: num(f.l10_stddev),
|
||
});
|
||
}
|
||
return { rows: out, clamped };
|
||
}
|
||
|
||
const ABLATIONS = [
|
||
['no_opp', 'opponent (opp_rank_stat, ±0.03)'],
|
||
['no_home', 'home/away (±0.015)'],
|
||
['no_consistency', 'consistency pull (cv>0.40 → toward 0.50)'],
|
||
['no_adjustments', 'ALL THREE adjustments (leaves base+recency)'],
|
||
];
|
||
|
||
const UNUSED = ['l5_avg', 'l10_avg', 'l20_avg', 'ab_per_game', 'recent_ab_per_game',
|
||
'opportunity_drift', 'rest_days', 'game_count_in_7d', 'park_h', 'park_hr', 'park_r',
|
||
'weather_temp_f', 'weather_wind_mph', 'weather_precip'];
|
||
const USED = ['opp_rank_stat', 'home_away', 'l10_stddev'];
|
||
|
||
function ablateStat(rows, label) {
|
||
const ys = rows.map((r) => r.won);
|
||
const full = r4(corr(rows.map((r) => r.full), ys));
|
||
const abl = {};
|
||
for (const [key, name] of ABLATIONS) {
|
||
const bs = bootstrapDiff(rows, key, 'full');
|
||
abl[name] = {
|
||
resolution_without: r4(corr(rows.map((r) => r[key]), ys)),
|
||
// NEGATIVE delta = removing it HURT = the feature carries signal.
|
||
delta_from_removal: bs ? bs.point : null,
|
||
ci95: bs ? bs.ci95 : null,
|
||
carries_signal: bs ? (bs.ci_excludes_zero && bs.point < 0) : null,
|
||
};
|
||
}
|
||
return { stat: label, n: rows.length, base_rate: r4(ys.reduce((a, b) => a + b, 0) / ys.length), resolution_full: full, ablations: abl };
|
||
}
|
||
|
||
function residualStat(rows, label) {
|
||
const scan = (keys, seedBase) => {
|
||
const out = {};
|
||
keys.forEach((k, i) => {
|
||
const res = bootstrapCorr(rows, k, 20260804 + i + seedBase);
|
||
out[k] = res;
|
||
});
|
||
return out;
|
||
};
|
||
return {
|
||
stat: label,
|
||
n: rows.length,
|
||
unused_features: scan(UNUSED, 0),
|
||
used_features_control: scan(USED, 500),
|
||
};
|
||
}
|
||
|
||
async function main() {
|
||
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
|
||
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
|
||
const { rows, clamped } = build(await fetchAll(sb));
|
||
|
||
const counts = {};
|
||
for (const r of rows) counts[r.stat] = (counts[r.stat] || 0) + 1;
|
||
const stats = Object.entries(counts).filter(([, n]) => n >= 60).map(([s]) => s)
|
||
.sort((a, b) => counts[b] - counts[a]);
|
||
|
||
const perStat = stats.map((s) => ablateStat(rows.filter((r) => r.stat === s), s));
|
||
const residual = stats.map((s) => residualStat(rows.filter((r) => r.stat === s), s));
|
||
|
||
// ── ARCHETYPE ON TRIAL ────────────────────────────────────────────────
|
||
// The champion reads NO archetype feature at all, so it cannot be ablated out
|
||
// of it. The fair test is whether archetype explains what the champion GETS
|
||
// WRONG: if a given archetype's rows are systematically mispriced, archetype
|
||
// carries prop signal the model is missing (wrong IMPLEMENTATION). If every
|
||
// archetype's mean residual straddles zero, archetype carries no prop signal.
|
||
const archetypeTest = stats.map((st) => {
|
||
const rs = rows.filter((r) => r.stat === st && r.archetype);
|
||
const groups = {};
|
||
for (const r of rs) (groups[r.archetype] = groups[r.archetype] || []).push(r.residual);
|
||
const out = {};
|
||
for (const [name, vals] of Object.entries(groups)) {
|
||
if (vals.length < 40) continue;
|
||
const rnd = makeRnd(20260805);
|
||
const means = [];
|
||
for (let it = 0; it < 3000; it += 1) {
|
||
let sum = 0;
|
||
for (let i = 0; i < vals.length; i += 1) sum += vals[Math.floor(rnd() * vals.length)];
|
||
means.push(sum / vals.length);
|
||
}
|
||
means.sort((a, b) => a - b);
|
||
const ci = [r4(means[Math.floor(0.025 * (means.length - 1))]), r4(means[Math.floor(0.975 * (means.length - 1))])];
|
||
out[name] = {
|
||
n: vals.length,
|
||
mean_residual: r4(vals.reduce((a, b) => a + b, 0) / vals.length),
|
||
ci95: ci,
|
||
systematically_mispriced: ci[0] > 0 || ci[1] < 0,
|
||
};
|
||
}
|
||
return { stat: st, archetypes: out };
|
||
});
|
||
|
||
// Any unused feature with a CI excluding zero, anywhere → a missing-feature lead.
|
||
const leads = [];
|
||
for (const rs of residual) {
|
||
for (const [k, v] of Object.entries(rs.unused_features)) {
|
||
if (v.ci_excludes_zero) leads.push({ stat: rs.stat, feature: k, corr: v.corr, ci95: v.ci95, n: v.n });
|
||
}
|
||
}
|
||
|
||
console.log(JSON.stringify({
|
||
measurement: 'EXACT ANALYTIC ABLATION of the champion, on the REPAIRED settled set. No refit, no re-fetch, no lookahead.',
|
||
limits: {
|
||
base_recency_not_separable: 'stored features hold AVERAGES, not frequencies-over-line; reported as one block',
|
||
clamped_rows_excluded: clamped,
|
||
},
|
||
total_rows: rows.length,
|
||
per_stat_ablation: perStat,
|
||
residual_signal_test: residual,
|
||
archetype_test: archetypeTest,
|
||
multiple_comparisons_note: 'The residual scan runs 14 unused features x 5 stats = 70 tests at alpha .05, so ~3-4 CI-excludes-zero results are EXPECTED BY CHANCE. Treat a single hit as noise; only a feature repeating across independent stats is evidence.',
|
||
missing_feature_leads: leads,
|
||
}, null, 2));
|
||
process.exit(0);
|
||
}
|
||
|
||
if (require.main === module) {
|
||
main().catch((e) => { console.error(e); process.exit(1); });
|
||
}
|
||
|
||
// Exported so the read-integrity harness measures THE REAL FUNCTION.
|
||
module.exports = { READS, READS_LEDGER };
|