Files
vyndr/scripts/prove-hit-factors.js
T
builtbykev f61ec6b391 Read integrity, as-of context, and the shadow matchup resolve (A1-A7)
Seven orders of measurement-first repair. The served grade does not move.

A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS:
410-617 of 2,490 duplicated with an equal number never returned, while
rows.length matched the server exactly. safePaginate orders on a real unique
key, verifies the tuple at runtime, and THROWS on a query error instead of
treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn
through that reader, and defense_by_direction's distinct-n was likely below
the gate floor all along.

A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from
pg_index (the context tables are dated-composite and had no single unique
column). The unordered helper is deleted, not parked.

A3 — ledgerService and retentionService defaulted the SAME env var to
DIFFERENT versions, so no ledger row ever carried the marker eligibility
requires. One source now. model_snapshots settlement moved onto the cron:
15,484 -> 28,894 settled, repaired-champion 0 -> 7,556.

A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no
row at-or-before the date means the factor does not apply, never the nearest
row. Live path unchanged, proven 400/400 on real rows.

A5 — factor_inputs freezes what the factor READ, never the multiplier, so an
audit can recompute and check. It also recorded the finding: the three hits
factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by
the resolver and written by nothing.

A6/A7 — matchupKeys resolves those keys from the posted lineup plus the
schedule's probable pitchers, and fires the factors into a SHADOW freeze:
248 fires on 308 props, 245 of which would move the grade. The served
forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test
that decides whether they ever go live.

Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay
withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity
harness 34/34.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-11 22:49:56 -04:00

342 lines
16 KiB
JavaScript

#!/usr/bin/env node
'use strict';
/**
* prove-hit-factors — does the hit grade read tonight's game, or say "he's due"?
*
* Each factor is conditioned against the player's OWN base rate and put through
* the two-part gate: it must MOVE the prediction and the moved prediction must
* be MORE ACCURATE out-of-sample. Movement alone is THEATER — a grade that
* swings on park and platoon looks like it read the matchup, and a user cannot
* tell the difference from outside.
*
* The baseline is deliberately the honest null this order describes: the
* player's base rate, i.e. "he's due" with no reading of tonight at all. A
* factor earns its place only by beating that.
*
* SUPABASE_URL=... node scripts/prove-hit-factors.js
*/
require('dotenv').config();
const { createClient } = require('@supabase/supabase-js');
const fg = require('../src/services/model/factorGate');
const sk = require('../src/services/model/skillProjection');
const tl = require('../src/services/model/testLedger');
const mlb = require('../src/services/adapters/mlbStatsAdapter');
const { knownNumber, knownRate } = require('../src/utils/known');
const { nameKey } = require('../src/utils/playerName');
const sd = require('../src/services/model/sprayDefense');
const pss = require('../src/services/model/platoonSeverity');
const { paginate } = require('../src/utils/safePaginate');
const { uniqueKeyFor } = require('../src/utils/tableKeys');
const SB_URL = process.env.SUPABASE_URL;
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
const PAGE = 1000;
const ARCHS = (process.env.HF_ARCHETYPES || 'BOMBER,GHOST,ALL').split(',');
// ── FIX A2 (2026-08-09) — THE SAFE WALK ───────────────────────────────────
// `page()` above walks with no ORDER BY. Measured on production, that returned
// the correct row COUNT and the wrong ROWS: up to 33.6% of a read came back
// twice while an equal share never came back at all. `pageSafe` routes the same
// call through `src/utils/safePaginate`, which orders on a UNIQUE key on every
// page, verifies uniqueness at runtime, and THROWS on a query error instead of
// treating it as end-of-data.
//
// `page()` SURVIVES only for the context tables (statcast_aggregates,
// batter_spray, team_defense, platoon_splits, park_dimensions, game_context...).
// Those have COMPOSITE primary keys with no single unique column, so
// safePaginate cannot express them. They measure 0% corruption today; making
// them safe needs a composite-key ordering the helper does not yet have. Do not
// use `page()` for ledger_entries or model_snapshots.
async function pageSafe(sb, table, select, apply, key = uniqueKeyFor(table)) {
return paginate(() => apply(sb.from(table).select(select)),
{ key, pageSize: PAGE, label: `${table}` });
}
/**
* THE MEASURED READS, named once.
*
* `main()` calls these and so does the read-integrity harness, so a PASS there
* is a statement about the code that actually runs. A registry that restated
* the query would verify the restatement instead.
*/
const READS = {
statcast: (sb) => pageSafe(sb, 'statcast_aggregates', '*', (q) => q.eq('sport', 'mlb')),
spray: (sb) => pageSafe(sb, 'batter_spray', '*', (q) => q.eq('sport', 'mlb')),
platoon: (sb) => pageSafe(sb, 'platoon_splits', '*', (q) => q.eq('sport', 'mlb')),
defense: (sb) => pageSafe(sb, 'team_defense', '*', (q) => q.eq('sport', 'mlb')),
snaps: (sb) => pageSafe(sb, 'model_snapshots', 'id, player_key, game_date, archetype, stat',
(q) => q.eq('sport', 'mlb').eq('stat', 'hits').not('archetype', 'is', null)),
ledger: (sb) => pageSafe(sb, 'ledger_entries',
'id, game_id, player_key, player_name, line, side, outcome, game_date, p_win, quarantine_reason, env_park_base',
(q) => q.eq('sport', 'mlb').is('user_id', null).eq('stat', 'hits')
.in('outcome', ['hit', 'miss']).not('p_win', 'is', null)),
};
/**
* THE FACTORS. Each returns a MULTIPLIER on the base rate, or null when the
* input is absent — an absent factor must leave the baseline untouched rather
* than nudge it toward some default.
*/
const FACTORS = [
{
key: 'defense_by_direction',
needs: ['spray_multiplier'],
entity: (r) => `${r.player_key}|${r.opp}`,
mechanism: 'CAUSALLY-CORRECT DEFENCE. Where the hitter puts the ball (pull/straight/oppo x ground/air) crossed with the OAA of the fielders actually standing in those zones, joined by handedness. Team-average failed the gate because it averages in five fielders who will never touch his ball.',
apply: (r) => r.spray_multiplier,
},
{
key: 'defense',
needs: ['team_defense'],
entity: (r) => r.opp,
mechanism: 'A ball in play becomes a hit or an out partly by who is standing behind the pitcher. Should matter most where contact stays in the park.',
// More outs converted above average -> fewer hits.
apply: (r) => 1 - Math.max(-0.12, Math.min(0.12, r.team_defense / 250)),
},
{
key: 'pitcher_contact_profile',
needs: ['pitcher_hard_hit_allowed'],
entity: (r) => r.starter_id,
mechanism: 'A contact-allowing arm concedes better contact than a bat-misser; hit probability should follow the quality of contact he permits.',
apply: (r) => 1 + Math.max(-0.15, Math.min(0.15, (r.pitcher_hard_hit_allowed - 0.389) * 1.2)),
},
{
key: 'park_hits',
needs: ['park_factor'],
entity: (r) => r.park_factor,
mechanism: 'Some parks turn outs into hits without producing runs — big outfields, high walls, deep gaps.',
apply: (r) => r.park_factor,
caveat: 'STAT_BASE maps hits -> run_base, so this is a RUN factor standing in for a HITS factor. A park that converts outs to hits without scoring is invisible to it.',
},
{
key: 'platoon_severity',
needs: ['platoon_severity_mult'],
entity: (r) => r.player_key,
mechanism: "CAUSALLY-CORRECT PLATOON. The advantage is worth only what THIS hitter's measured split is worth, shrunk toward league by the smaller side's PA and refused outright below a floor. Flat handedness applies the same boost to a 63-point split and to none.",
apply: (r) => r.platoon_severity_mult,
},
{
key: 'platoon',
needs: ['platoon_edge'],
entity: (r) => r.player_key,
mechanism: 'Handedness advantage — a hitter facing the opposite hand sees the ball better and hits it harder.',
apply: (r) => (r.platoon_edge > 0 ? 1.06 : 0.96),
},
];
async function main() {
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
const statcast = await pageSafe(sb, 'statcast_aggregates', '*', (q) => q.eq('sport', 'mlb'));
const batters = new Map(); const pitchersById = new Map();
for (const r of statcast) {
const prof = sk.fromStatcastRow(r);
if (r.role === 'pitcher' && r.source_id != null) pitchersById.set(Number(r.source_id), prof);
if (r.role === 'batter' && r.player_key) batters.set(r.player_key, prof);
}
const sprayRows = await pageSafe(sb, 'batter_spray', '*', (q) => q.eq('sport', 'mlb'));
const sprayByKey = new Map();
for (const r of sprayRows) {
if (!r.player_key) continue;
const prev = sprayByKey.get(r.player_key);
if (!prev || String(r.as_of_date) > String(prev.as_of_date)) sprayByKey.set(r.player_key, r);
}
const platRows = await pageSafe(sb, 'platoon_splits', '*', (q) => q.eq('sport', 'mlb'));
const platByKey = new Map();
for (const r of platRows) {
if (!r.player_key) continue;
const prev = platByKey.get(r.player_key);
if (!prev || String(r.as_of_date) > String(prev.as_of_date)) platByKey.set(r.player_key, r);
}
const defRows = await pageSafe(sb, 'team_defense', '*', (q) => q.eq('sport', 'mlb'));
const defByTeam = new Map();
for (const d of defRows) defByTeam.set(d.team, d);
const snaps = await READS.snaps(sb);
const archOf = new Map();
for (const s of snaps) archOf.set(`${s.player_key}|${s.game_date}`, s.archetype);
const led = await READS.ledger(sb);
const clean = led.filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book'));
// Opponent faced, from each hitter's own game log.
const names = new Map();
for (const r of clean) if (!names.has(r.player_key)) names.set(r.player_key, r.player_name);
const oppBy = new Map(); const startersBy = new Map();
const dates = [...new Set(clean.map((r) => r.game_date))].sort();
for (const d of dates) {
try {
const games = await mlb.getScheduleWithPitchers(d);
for (const g of games) {
if (!g.home || !g.away) continue;
if (g.home.probablePitcher) startersBy.set(`${d}|OPP:${g.home.team}`, g.home.probablePitcher.id);
if (g.away.probablePitcher) startersBy.set(`${d}|OPP:${g.away.team}`, g.away.probablePitcher.id);
}
} catch { /* absent slate */ }
}
for (const [key, name] of names) {
try {
const found = await mlb.searchPlayer(name);
if (!found || !found.id) continue;
const log = await mlb.getPlayerGameLog(found.id);
for (const g of log || []) if (g && g.date && g.opponent) oppBy.set(`${key}|${String(g.date).slice(0, 10)}`, g.opponent);
} catch { /* no log */ }
}
// Per-player base rate — the honest null: "he's due", no reading of tonight.
const byPlayer = new Map();
for (const r of clean) {
const cur = byPlayer.get(r.player_key) || { n: 0, w: 0 };
cur.n += 1; cur.w += r.outcome === 'hit' ? 1 : 0;
byPlayer.set(r.player_key, cur);
}
const loss = { no_batter_profile: 0, thin_base_rate: 0, no_opponent: 0, no_pitcher: 0, kept: 0 };
const rows = [];
for (const r of clean) {
const bat = batters.get(r.player_key);
const bp = byPlayer.get(r.player_key);
if (!bat) loss.no_batter_profile += 1;
if (!bp || bp.n < 3) { loss.thin_base_rate += 1; continue; }
// Leave-one-out so a row never contributes to its own baseline.
const baseline = (bp.w - (r.outcome === 'hit' ? 1 : 0)) / (bp.n - 1);
const faced = oppBy.get(`${r.player_key}|${r.game_date}`) || null;
const nick = faced ? String(faced).split(' ').pop() : null;
const def = faced ? (defByTeam.get(faced) || defByTeam.get(nick)) : null;
if (!faced) loss.no_opponent += 1;
const starterId = faced ? startersBy.get(`${r.game_date}|OPP:${faced}`) : null;
const pit = starterId != null ? pitchersById.get(Number(starterId)) : null;
if (faced && !pit) loss.no_pitcher += 1;
loss.kept += 1;
rows.push({
id: r.id,
// Errors are correlated WITHIN a game — shared starter, park, weather and
// the game's own randomness — so the interval must be clustered on it.
// Three of these factors (pitcher profile, team defence, park) are also
// CONSTANT across every hitter facing that starter, which makes row
// resampling straightforwardly wrong for them.
cluster: r.game_id,
opp: faced,
starter_id: starterId != null ? Number(starterId) : null,
player_key: r.player_key,
archetype: archOf.get(`${r.player_key}|${r.game_date}`) || null,
won: r.outcome === 'hit' ? 1 : 0,
baseline,
team_defense: def ? knownNumber(def.oaa_sum) : null,
pitcher_hard_hit_allowed: pit ? knownRate(pit.hard_hit_pct) : null,
park_factor: knownNumber(r.env_park_base),
platoon_severity_mult: (() => {
const sp = platByKey.get(r.player_key);
if (!sp || !bat || !bat.bats || !pit || !pit.throws) return null;
const out = pss.platoonRead({
splits: {
vl: { pa: sp.vl_pa, atBats: sp.vl_ab, hits: sp.vl_hits },
vr: { pa: sp.vr_pa, atBats: sp.vr_ab, hits: sp.vr_hits },
},
bats: bat.bats, throws: pit.throws,
});
return out && out.readable ? out.multiplier : null;
})(),
spray_multiplier: (() => {
const sp = sprayByKey.get(r.player_key);
const posOaa = def && def.position_oaa ? def.position_oaa : null;
if (!sp || !posOaa || !bat || !bat.bats) return null;
const out = sd.sprayDefenseMultiplier({ spray: sp, bats: bat.bats, positionOaa: posOaa });
return out ? out.multiplier : null;
})(),
platoon_edge: (bat && pit && bat.bats && pit.throws)
? (String(bat.bats)[0] !== String(pit.throws)[0] ? 1 : -1) : null,
});
}
// Cumulative Bonferroni across the programme lifetime.
const store = tl.supabaseStore(sb);
const mc = await tl.recordAndCount(store, FACTORS.flatMap((f) =>
ARCHS.map((a) => ({ sport: 'mlb', stat: 'hits', archetype: a === 'ALL' ? null : a, interaction: `factor:${f.key}`, target: 'outcome' }))));
// STEP 1 — FULL-HISTORY SAMPLE AUDIT PER SLOT, before any gating.
const audit = [];
for (const f of FACTORS) {
for (const arch of ARCHS) {
const slot = arch === 'ALL' ? rows : rows.filter((r) => String(r.archetype || '').toUpperCase() === arch);
const usable = slot.filter((r) => f.needs.every((k) => knownNumber(r[k]) !== null));
audit.push({
factor: f.key,
archetype: arch,
rows: usable.length,
games: new Set(usable.map((r) => r.cluster).filter(Boolean)).size,
players: new Set(usable.map((r) => r.player_key)).size,
});
}
}
const results = [];
for (const arch of ARCHS) {
const slot = arch === 'ALL' ? rows : rows.filter((r) => String(r.archetype || '').toUpperCase() === arch);
for (const f of FACTORS) {
const usable = slot.filter((r) => f.needs.every((k) => knownNumber(r[k]) !== null));
// A park effect is replicated across PARKS, not across games: 619 rows in
// 45 games still only ever saw ~23 ballparks, and unmodelled park
// heterogeneity is confounded with the very thing being estimated. So the
// cluster is the COARSER of the game and the entity the treatment rides on.
const ents = f.entity ? new Set(usable.map((r) => String(f.entity(r)))) : null;
const games = new Set(usable.map((r) => String(r.cluster)));
const useEntity = ents && ents.size < games.size;
const paired = usable.map((r) => {
const mult = f.apply(r);
const cond = mult === null ? null : Math.min(0.99, Math.max(0.01, r.baseline * mult));
return {
baseline: r.baseline,
conditioned: cond,
won: r.won,
cluster: useEntity ? `e:${f.entity(r)}` : r.cluster,
};
});
const v = fg.adjudicate(paired, {
factor: f.key, archetype: arch, stat: 'hits',
cumulativeTests: mc.cumulative_tests, // native cumulative correction
});
results.push({
archetype: arch, factor: f.key, n: v.movement.n,
clusters: v.improvement ? v.improvement.effective_n : null,
cluster_unit: useEntity ? 'treatment_entity' : 'game',
distinct_games: games.size,
distinct_entities: ents ? ents.size : null,
mean_abs_shift: v.movement.mean_abs_shift,
brier_delta: v.improvement ? v.improvement.brier_delta : null,
ci: v.improvement ? v.improvement.ci : null,
ci_level: v.improvement ? v.improvement.ci_level : null,
verdict: v.verdict,
reason: v.reason,
...(f.caveat ? { input_caveat: f.caveat } : {}),
});
}
}
console.log(JSON.stringify({
baseline: "each row scored against the player's OWN leave-one-out base rate — the honest 'he's due' null",
total_rows: rows.length,
slot_audit: audit,
clean_settled_rows_available: clean.length,
row_loss: loss,
cumulative_bonferroni: mc,
gate: 'a factor must MOVE the prediction AND improve out-of-sample Brier; movement alone is THEATER',
results,
proven: results.filter((r) => r.verdict === 'PROVES'),
theater: results.filter((r) => r.verdict === 'THEATER'),
}, null, 2));
process.exit(0);
}
if (require.main === module) {
main().catch((e) => { console.error(e); process.exit(1); });
}
// Exported so the read-integrity harness measures THE REAL FUNCTION.
module.exports = { READS };