f61ec6b391
Seven orders of measurement-first repair. The served grade does not move. A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS: 410-617 of 2,490 duplicated with an equal number never returned, while rows.length matched the server exactly. safePaginate orders on a real unique key, verifies the tuple at runtime, and THROWS on a query error instead of treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn through that reader, and defense_by_direction's distinct-n was likely below the gate floor all along. A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from pg_index (the context tables are dated-composite and had no single unique column). The unordered helper is deleted, not parked. A3 — ledgerService and retentionService defaulted the SAME env var to DIFFERENT versions, so no ledger row ever carried the marker eligibility requires. One source now. model_snapshots settlement moved onto the cron: 15,484 -> 28,894 settled, repaired-champion 0 -> 7,556. A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no row at-or-before the date means the factor does not apply, never the nearest row. Live path unchanged, proven 400/400 on real rows. A5 — factor_inputs freezes what the factor READ, never the multiplier, so an audit can recompute and check. It also recorded the finding: the three hits factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by the resolver and written by nothing. A6/A7 — matchupKeys resolves those keys from the posted lineup plus the schedule's probable pitchers, and fires the factors into a SHADOW freeze: 248 fires on 308 props, 245 of which would move the grade. The served forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test that decides whether they ever go live. Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity harness 34/34. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
299 lines
14 KiB
JavaScript
299 lines
14 KiB
JavaScript
#!/usr/bin/env node
|
|
'use strict';
|
|
|
|
/**
|
|
* skill-v1-stagea — DOES THE WINDSHIELD BEAT THE REAR-VIEW MIRROR?
|
|
*
|
|
* Stage A's only question: on settled HITS props, does an archetype-selected,
|
|
* skill-based forward projection call the LISTED LINE better than the frequency
|
|
* counter that is the reigning champion? If it does not, it is not real yet and
|
|
* it does not get promoted. That is the whole test.
|
|
*
|
|
* ── WHY THIS IS GENUINELY OUT-OF-SAMPLE ──────────────────────────────────
|
|
* `statcast_aggregates` was last refreshed 2026-07-21 (the nightly job was
|
|
* unreachable code until this session — see snapshotScheduler). Settled hits
|
|
* rows run 2026-07-23 onward. So the skill profiles this model reads were
|
|
* frozen BEFORE every game it is asked to predict. The staleness that was a bug
|
|
* for production is, for this one measurement, a clean point-in-time snapshot.
|
|
* Rows on or before the freeze date are EXCLUDED so no profile can contain the
|
|
* game it is predicting.
|
|
*
|
|
* The opposing starter comes from the statsapi schedule for that date, and the
|
|
* pitcher's skill profile from the same frozen aggregate table.
|
|
*
|
|
* ── THE BAR (identical to the one that refuted hits-v1) ──────────────────
|
|
* - hits rows only, direction-aligned to the graded side
|
|
* - matched rows only: champion and challenger scored on the SAME props
|
|
* - paired bootstrap, deterministic seed, CI on the DIFFERENCE
|
|
* - PROMOTE only if the CI excludes zero on the good side
|
|
*
|
|
* ── DISCIPLINE 4, MEASURED, NOT ASSUMED ──────────────────────────────────
|
|
* Selectivity is reported, not claimed: accuracy is broken out by how confident
|
|
* the model is, so "right 57% on the 8 you're sure of" is a number rather than a
|
|
* slogan. LIFT over the naive base rate is reported beside it, because being
|
|
* right about obvious chalk is not signal.
|
|
*
|
|
* SUPABASE_URL=... node scripts/skill-v1-stagea.js
|
|
*/
|
|
|
|
require('dotenv').config();
|
|
const { createClient } = require('@supabase/supabase-js');
|
|
const sk = require('../src/services/model/skillProjection');
|
|
const reg = require('../src/services/model/featureRegistry');
|
|
const mlb = require('../src/services/adapters/mlbStatsAdapter');
|
|
const { nameKey } = require('../src/utils/playerName');
|
|
const { knownRate } = require('../src/utils/known');
|
|
const { paginate } = require('../src/utils/safePaginate');
|
|
const { uniqueKeyFor } = require('../src/utils/tableKeys');
|
|
|
|
/** Team games played by the 2026-07-21 profile freeze — turns season PA into PA/game. */
|
|
const GAMES_SO_FAR = Number(process.env.STAGEA_GAMES_SO_FAR || 103);
|
|
|
|
const SB_URL = process.env.SUPABASE_URL;
|
|
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
|
|
const PAGE = 1000;
|
|
|
|
function corr(xs, ys) {
|
|
const n = xs.length;
|
|
if (n < 3) return null;
|
|
const mx = xs.reduce((a, b) => a + b, 0) / n;
|
|
const my = ys.reduce((a, b) => a + b, 0) / n;
|
|
let sxy = 0; let sxx = 0; let syy = 0;
|
|
for (let i = 0; i < n; i += 1) {
|
|
const dx = xs[i] - mx; const dy = ys[i] - my;
|
|
sxy += dx * dy; sxx += dx * dx; syy += dy * dy;
|
|
}
|
|
if (sxx <= 0 || syy <= 0) return null;
|
|
return sxy / Math.sqrt(sxx * syy);
|
|
}
|
|
const r4 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 10000) / 10000);
|
|
const mean = (a) => (a.length ? a.reduce((x, y) => x + y, 0) / a.length : null);
|
|
const brier = (ps, ys) => (ps.length ? ps.reduce((s, p, i) => s + (p - ys[i]) ** 2, 0) / ps.length : null);
|
|
|
|
function makeRnd(seed) {
|
|
let s = seed >>> 0;
|
|
return () => { s ^= s << 13; s >>>= 0; s ^= s >>> 17; s ^= s << 5; s >>>= 0; return s / 4294967296; };
|
|
}
|
|
|
|
function bootstrapDiff(rows, keyA, keyB, iters = 4000, seed = 20260803) {
|
|
if (rows.length < 30) return null;
|
|
const rnd = makeRnd(seed);
|
|
const n = rows.length;
|
|
const diffs = [];
|
|
for (let it = 0; it < iters; it += 1) {
|
|
const ys = []; const a = []; const b = [];
|
|
for (let i = 0; i < n; i += 1) {
|
|
const r = rows[Math.floor(rnd() * n)];
|
|
ys.push(r.won); a.push(r[keyA]); b.push(r[keyB]);
|
|
}
|
|
const ca = corr(a, ys); const cb = corr(b, ys);
|
|
if (ca == null || cb == null) continue;
|
|
diffs.push(ca - cb);
|
|
}
|
|
if (diffs.length < 100) return null;
|
|
diffs.sort((x, y) => x - y);
|
|
const q = (p) => r4(diffs[Math.floor(p * (diffs.length - 1))]);
|
|
const ci = [q(0.025), q(0.975)];
|
|
return {
|
|
point: r4(corr(rows.map((r) => r[keyA]), rows.map((r) => r.won))
|
|
- corr(rows.map((r) => r[keyB]), rows.map((r) => r.won))),
|
|
ci95: ci,
|
|
ci_excludes_zero: ci[0] > 0 || ci[1] < 0,
|
|
};
|
|
}
|
|
|
|
// ── THE SAFE WALK (Fix A2/A2b) ────────────────────────────────────────────
|
|
// This script used to walk pages with `.range()` and NO ORDER BY. Measured on
|
|
// production, that returned the correct row COUNT and the wrong ROWS: up to
|
|
// 33.6% of a read came back twice while an equal share never came back at all,
|
|
// so `rows.length` looked perfect while a fifth of the sample was missing.
|
|
//
|
|
// `pageSafe` routes every read through `src/utils/safePaginate`: a stable ORDER
|
|
// BY on the table's real UNIQUE key — single OR composite, looked up from
|
|
// `src/utils/tableKeys` rather than assumed — a runtime tuple-uniqueness check,
|
|
// and a THROWN error instead of a silent stop. The old unordered helper is gone
|
|
// rather than left beside it, because a dead broken helper is an invitation.
|
|
async function pageSafe(sb, table, select, apply, key = uniqueKeyFor(table)) {
|
|
return paginate(() => apply(sb.from(table).select(select)),
|
|
{ key, pageSize: PAGE, label: `${table}` });
|
|
}
|
|
|
|
/** THE MEASURED READS — main() and the harness call the same functions. */
|
|
const READS = {
|
|
ledger: (sb) => pageSafe(sb, 'ledger_entries',
|
|
'id, player_key, player_name, line, side, outcome, game_date, p_win, quarantine_reason',
|
|
(q) => q.eq('sport', 'mlb').is('user_id', null).eq('stat', 'hits')
|
|
.in('outcome', ['hit', 'miss'])),
|
|
};
|
|
|
|
/**
|
|
* `date|team` → the starter that team FACED.
|
|
*
|
|
* Built from the statsapi schedule: a team faces the OTHER side's probable.
|
|
*/
|
|
async function opposingStarters(dates) {
|
|
const byDateTeam = new Map();
|
|
for (const d of dates) {
|
|
let games = [];
|
|
try { games = await mlb.getScheduleWithPitchers(d); } catch { games = []; }
|
|
for (const g of games) {
|
|
if (!g.home || !g.away) continue;
|
|
if (g.away.probablePitcher) byDateTeam.set(`${d}|${g.home.team}`, g.away.probablePitcher.id);
|
|
if (g.home.probablePitcher) byDateTeam.set(`${d}|${g.away.team}`, g.home.probablePitcher.id);
|
|
// Also key by the PITCHING team, so "who did team X send out" is directly
|
|
// answerable from the opponent name a game log gives us.
|
|
if (g.home.probablePitcher) byDateTeam.set(`${d}|OPP:${g.home.team}`, g.home.probablePitcher.id);
|
|
if (g.away.probablePitcher) byDateTeam.set(`${d}|OPP:${g.away.team}`, g.away.probablePitcher.id);
|
|
}
|
|
}
|
|
return byDateTeam;
|
|
}
|
|
|
|
/**
|
|
* `playerKey|date` → the OPPONENT team that player faced.
|
|
*
|
|
* THE LEDGER CANNOT ANSWER THIS: `team`/`opponent` are NULL on 575 of 576 rows
|
|
* in this window, which is why the first run resolved a pitcher for exactly ONE
|
|
* row and silently measured a batter-profile-only model instead of the matchup
|
|
* model it claimed to test. The player's own statsapi game log names the
|
|
* opponent for the exact date, so it is both authoritative and point-in-time
|
|
* safe (a completed game's opponent is not a forecast).
|
|
*/
|
|
async function opponentByPlayerDate(players) {
|
|
const map = new Map();
|
|
for (const [key, name] of players) {
|
|
try {
|
|
const found = await mlb.searchPlayer(name);
|
|
if (!found || !found.id) continue;
|
|
const log = await mlb.getPlayerGameLog(found.id);
|
|
for (const g of log || []) {
|
|
if (g && g.date && g.opponent) map.set(`${key}|${String(g.date).slice(0, 10)}`, g.opponent);
|
|
}
|
|
} catch { /* a missing log just means no pitcher for those rows */ }
|
|
}
|
|
return map;
|
|
}
|
|
|
|
async function main() {
|
|
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
|
|
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
|
|
|
|
// Frozen skill profiles — by name (batters) and by source_id (pitchers).
|
|
const statcast = await pageSafe(sb, 'statcast_aggregates', '*', (q) => q.eq('sport', 'mlb'));
|
|
const freeze = statcast.reduce((mx, r) => (String(r.updated_at) > mx ? String(r.updated_at) : mx), '');
|
|
const freezeDate = freeze.slice(0, 10);
|
|
const batters = new Map();
|
|
const pitchersById = new Map();
|
|
for (const r of statcast) {
|
|
// UNITS: statcast_aggregates stores percentages (0-100). Convert ONCE, here.
|
|
if (r.role === 'pitcher' && r.source_id != null) pitchersById.set(Number(r.source_id), sk.fromStatcastRow(r));
|
|
if (r.player_key && r.role === 'batter') {
|
|
const prev = batters.get(r.player_key);
|
|
const size = Number(r.sample_pa || 0);
|
|
if (!prev || size > Number(prev.rawPa || 0)) {
|
|
batters.set(r.player_key, Object.assign(sk.fromStatcastRow(r), { rawPa: size, archetype: null }));
|
|
}
|
|
}
|
|
}
|
|
|
|
const led = await pageSafe(sb, 'ledger_entries',
|
|
'id, player_key, player_name, stat, line, side, outcome, game_date, p_win, team, opponent, quarantine_reason',
|
|
(q) => q.eq('sport', 'mlb').is('user_id', null).eq('stat', 'hits')
|
|
.in('outcome', ['hit', 'miss']).not('p_win', 'is', null));
|
|
|
|
const clean = led.filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book')
|
|
// STRICTLY AFTER the profile freeze — no game may be inside its own inputs.
|
|
&& String(r.game_date) > freezeDate);
|
|
|
|
const dates = [...new Set(clean.map((r) => r.game_date))].sort();
|
|
const starters = await opposingStarters(dates);
|
|
const players = new Map();
|
|
for (const r of clean) if (!players.has(r.player_key)) players.set(r.player_key, r.player_name);
|
|
const oppByPlayerDate = await opponentByPlayerDate(players);
|
|
|
|
const allowed = reg.candidateFeatures('mlb'); // challenger-first: candidates measured, never served
|
|
const rows = [];
|
|
const drops = {};
|
|
const drop = (k) => { drops[k] = (drops[k] || 0) + 1; };
|
|
let withPitcher = 0;
|
|
|
|
for (const r of clean) {
|
|
const bat = batters.get(r.player_key);
|
|
if (!bat) { drop('no_batter_profile'); continue; }
|
|
// The team he FACED that day, from his own game log. `starters` is keyed by
|
|
// the team doing the facing, so look up his own team — which is the
|
|
// opponent's opponent. Resolve via the game log's opponent name and take the
|
|
// schedule entry for the OTHER side.
|
|
const facedTeam = oppByPlayerDate.get(`${r.player_key}|${r.game_date}`) || null;
|
|
const starterId = facedTeam ? starters.get(`${r.game_date}|OPP:${facedTeam}`) : null;
|
|
const pit = starterId != null ? pitchersById.get(Number(starterId)) : null;
|
|
if (pit) withPitcher += 1;
|
|
|
|
// Opportunity: the batter's own season PA per game, from the frozen profile.
|
|
// Opportunity: season PA spread over the games played so far this season.
|
|
// GAMES_SO_FAR is the frozen-profile era's team game count, so PA/game is a
|
|
// real per-game rate rather than an arbitrary divisor.
|
|
const pa = knownRate(bat.rawPa);
|
|
const expectedPa = pa && pa > 0 ? Math.min(5.2, Math.max(2.0, pa / GAMES_SO_FAR)) : null;
|
|
|
|
const out = sk.projectSkill({
|
|
batter: bat, pitcher: pit, park: 1,
|
|
archetype: bat.archetype || null,
|
|
statType: 'hits', line: Number(r.line),
|
|
expectedPa, allowed,
|
|
});
|
|
if (!out) { drop('projection_refused'); continue; }
|
|
|
|
const under = String(r.side).toLowerCase() === 'under';
|
|
rows.push({
|
|
won: r.outcome === 'hit' ? 1 : 0,
|
|
champ: Number(r.p_win),
|
|
skill: under ? 1 - out.p_over_line : out.p_over_line,
|
|
line: Number(r.line),
|
|
had_pitcher: !!pit,
|
|
});
|
|
}
|
|
|
|
const ys = rows.map((r) => r.won);
|
|
const base = mean(ys);
|
|
const bs = bootstrapDiff(rows, 'skill', 'champ');
|
|
|
|
// DISCIPLINE 4 — selectivity, measured. Sorted by confidence in the graded
|
|
// side; report accuracy and LIFT over the naive base rate at each depth.
|
|
const byConf = [...rows].sort((a, b) => b.skill - a.skill);
|
|
const depths = [8, 15, 25, 50, 100].filter((d) => d <= byConf.length);
|
|
const selectivity = depths.map((d) => {
|
|
const top = byConf.slice(0, d);
|
|
const hit = mean(top.map((r) => r.won));
|
|
return { top_n: d, hit_rate: r4(hit), lift_over_base: r4(hit - base) };
|
|
});
|
|
|
|
console.log(JSON.stringify({
|
|
measurement: 'STAGE A — skill-v1 vs the frequency counter, out-of-sample on listed-line accuracy',
|
|
out_of_sample_guarantee: `skill profiles frozen ${freezeDate}; only rows with game_date > ${freezeDate} scored`,
|
|
registry: reg.summary('mlb'),
|
|
matched_rows: rows.length,
|
|
rows_with_opposing_pitcher: withPitcher,
|
|
pitcher_coverage_pct: rows.length ? r4(withPitcher / rows.length) : null,
|
|
dropped: drops,
|
|
base_rate: r4(base),
|
|
resolution: { skill_v1: r4(corr(rows.map((r) => r.skill), ys)), champion: r4(corr(rows.map((r) => r.champ), ys)) },
|
|
brier: { skill_v1: r4(brier(rows.map((r) => r.skill), ys)), champion: r4(brier(rows.map((r) => r.champ), ys)) },
|
|
mean_forecast: { skill_v1: r4(mean(rows.map((r) => r.skill))), champion: r4(mean(rows.map((r) => r.champ))) },
|
|
delta_vs_champion: bs,
|
|
verdict: !bs ? 'N-BLOCKED'
|
|
: (bs.ci_excludes_zero && bs.point > 0) ? 'BEATS THE COUNTER — promotable'
|
|
: (bs.ci_excludes_zero && bs.point < 0) ? 'LOSES to the counter — iterate, do not promote'
|
|
: 'INCONCLUSIVE — not proven, do not promote',
|
|
selectivity_discipline_4: selectivity,
|
|
}, null, 2));
|
|
process.exit(0);
|
|
}
|
|
|
|
if (require.main === module) {
|
|
main().catch((e) => { console.error(e); process.exit(1); });
|
|
}
|
|
|
|
// Exported so the read-integrity harness measures THE REAL FUNCTION.
|
|
module.exports = { READS };
|