Files
vyndr/scripts/stagea-gate-run.js
T
builtbykev f61ec6b391 Read integrity, as-of context, and the shadow matchup resolve (A1-A7)
Seven orders of measurement-first repair. The served grade does not move.

A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS:
410-617 of 2,490 duplicated with an equal number never returned, while
rows.length matched the server exactly. safePaginate orders on a real unique
key, verifies the tuple at runtime, and THROWS on a query error instead of
treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn
through that reader, and defense_by_direction's distinct-n was likely below
the gate floor all along.

A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from
pg_index (the context tables are dated-composite and had no single unique
column). The unordered helper is deleted, not parked.

A3 — ledgerService and retentionService defaulted the SAME env var to
DIFFERENT versions, so no ledger row ever carried the marker eligibility
requires. One source now. model_snapshots settlement moved onto the cron:
15,484 -> 28,894 settled, repaired-champion 0 -> 7,556.

A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no
row at-or-before the date means the factor does not apply, never the nearest
row. Live path unchanged, proven 400/400 on real rows.

A5 — factor_inputs freezes what the factor READ, never the multiplier, so an
audit can recompute and check. It also recorded the finding: the three hits
factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by
the resolver and written by nothing.

A6/A7 — matchupKeys resolves those keys from the posted lineup plus the
schedule's probable pitchers, and fires the factors into a SHADOW freeze:
248 fires on 308 props, 245 of which would move the grade. The served
forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test
that decides whether they ever go live.

Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay
withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity
harness 34/34.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-11 22:49:56 -04:00

266 lines
12 KiB
JavaScript

#!/usr/bin/env node
'use strict';
/**
* stagea-gate-run — RUN THE SKILL FEATURES THROUGH THE GATE, THEN THE COUNTER.
*
* The original sin was never that the challengers were badly built. It was that
* every one of them was measured WITHOUT a validation gate, so "it didn't work"
* and "it was never allowed to prove it works" were indistinguishable. This runs
* the gate that spec'd for exactly this (n>=500, |r|>=0.15, p<0.05, Bonferroni)
* over the real skill features, and only then does the head-to-head.
*
* THREE MEASUREMENTS, in the order that makes each one meaningful:
*
* 1. RAW SIGNAL — corr(feature, outcome). Does this skill input relate to
* whether the prop hit at all?
* 2. MARGINAL CONTRIBUTION — corr(feature, counter residual). This is the one
* that matters: a feature can correlate with the outcome purely because the
* counter already knows it. Only the part the counter MISSES is new
* information, and that is what earns a place. Both go through the gate.
* 3. HEAD-TO-HEAD — the value projection vs the live counter on listed-line
* accuracy, paired bootstrap, out-of-sample.
*
* OUT-OF-SAMPLE: skill profiles are the frozen 2026-07-21 aggregate; only games
* AFTER that date are scored, so no profile contains the game it predicts.
*
* BONFERRONI DENOMINATOR is the number of features tested in this sweep — not 1.
* Testing many and reporting the best without correction is how the S78 residual
* scan produced six "findings" when chance alone predicts three or four.
*
* SUPABASE_URL=... node scripts/stagea-gate-run.js
*/
require('dotenv').config();
const { createClient } = require('@supabase/supabase-js');
const cv = require('../src/services/model/correlateValidator');
const sk = require('../src/services/model/skillProjection');
const reg = require('../src/services/model/featureRegistry');
const mlb = require('../src/services/adapters/mlbStatsAdapter');
const { knownRate, knownNumber } = require('../src/utils/known');
const { paginate } = require('../src/utils/safePaginate');
const { uniqueKeyFor } = require('../src/utils/tableKeys');
const SB_URL = process.env.SUPABASE_URL;
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
const PAGE = 1000;
const GAMES_SO_FAR = Number(process.env.STAGEA_GAMES_SO_FAR || 103);
const r4 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 10000) / 10000);
const mean = (a) => (a.length ? a.reduce((x, y) => x + y, 0) / a.length : null);
const brier = (ps, ys) => (ps.length ? ps.reduce((s, p, i) => s + (p - ys[i]) ** 2, 0) / ps.length : null);
function makeRnd(seed) {
let s = seed >>> 0;
return () => { s ^= s << 13; s >>>= 0; s ^= s >>> 17; s ^= s << 5; s >>>= 0; return s / 4294967296; };
}
function corrOf(xs, ys) { return cv.pearson(xs, ys).r; }
function bootstrapDiff(rows, keyA, keyB, iters = 4000, seed = 20260803) {
if (rows.length < 30) return null;
const rnd = makeRnd(seed);
const n = rows.length;
const diffs = [];
for (let it = 0; it < iters; it += 1) {
const ys = []; const a = []; const b = [];
for (let i = 0; i < n; i += 1) {
const r = rows[Math.floor(rnd() * n)];
ys.push(r.won); a.push(r[keyA]); b.push(r[keyB]);
}
const ca = corrOf(a, ys); const cb = corrOf(b, ys);
if (ca == null || cb == null) continue;
diffs.push(ca - cb);
}
if (diffs.length < 100) return null;
diffs.sort((x, y) => x - y);
const q = (p) => r4(diffs[Math.floor(p * (diffs.length - 1))]);
const ci = [q(0.025), q(0.975)];
return {
point: r4(corrOf(rows.map((r) => r[keyA]), rows.map((r) => r.won))
- corrOf(rows.map((r) => r[keyB]), rows.map((r) => r.won))),
ci95: ci, ci_excludes_zero: ci[0] > 0 || ci[1] < 0,
};
}
// ── THE SAFE WALK (Fix A2/A2b) ────────────────────────────────────────────
// This script used to walk pages with `.range()` and NO ORDER BY. Measured on
// production, that returned the correct row COUNT and the wrong ROWS: up to
// 33.6% of a read came back twice while an equal share never came back at all,
// so `rows.length` looked perfect while a fifth of the sample was missing.
//
// `pageSafe` routes every read through `src/utils/safePaginate`: a stable ORDER
// BY on the table's real UNIQUE key — single OR composite, looked up from
// `src/utils/tableKeys` rather than assumed — a runtime tuple-uniqueness check,
// and a THROWN error instead of a silent stop. The old unordered helper is gone
// rather than left beside it, because a dead broken helper is an invitation.
async function pageSafe(sb, table, select, apply, key = uniqueKeyFor(table)) {
return paginate(() => apply(sb.from(table).select(select)),
{ key, pageSize: PAGE, label: `${table}` });
}
/** THE MEASURED READS — main() and the harness call the same functions. */
const READS = {
ledger: (sb) => pageSafe(sb, 'ledger_entries',
'id, player_key, player_name, stat, line, side, outcome, game_date, p_win, quarantine_reason',
(q) => q.eq('sport', 'mlb').is('user_id', null)),
};
async function opposingStarters(dates) {
const m = new Map();
for (const d of dates) {
let games = [];
try { games = await mlb.getScheduleWithPitchers(d); } catch { games = []; }
for (const g of games) {
if (!g.home || !g.away) continue;
if (g.home.probablePitcher) m.set(`${d}|OPP:${g.home.team}`, g.home.probablePitcher.id);
if (g.away.probablePitcher) m.set(`${d}|OPP:${g.away.team}`, g.away.probablePitcher.id);
}
}
return m;
}
/** `playerKey|date` → opponent faced. The ledger's team/opponent are NULL. */
async function opponentByPlayerDate(players) {
const map = new Map();
for (const [key, name] of players) {
try {
const found = await mlb.searchPlayer(name);
if (!found || !found.id) continue;
const log = await mlb.getPlayerGameLog(found.id);
for (const g of log || []) {
if (g && g.date && g.opponent) map.set(`${key}|${String(g.date).slice(0, 10)}`, g.opponent);
}
} catch { /* no log → no pitcher for those rows */ }
}
return map;
}
async function main() {
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
const statcast = await pageSafe(sb, 'statcast_aggregates', '*', (q) => q.eq('sport', 'mlb'));
const freezeDate = statcast.reduce((mx, r) => (String(r.updated_at) > mx ? String(r.updated_at) : mx), '').slice(0, 10);
const batters = new Map(); const pitchersById = new Map();
for (const r of statcast) {
if (r.role === 'pitcher' && r.source_id != null) pitchersById.set(Number(r.source_id), sk.fromStatcastRow(r));
if (r.player_key && r.role === 'batter') {
const prev = batters.get(r.player_key);
if (!prev || Number(r.sample_pa || 0) > Number(prev.rawPa || 0)) {
batters.set(r.player_key, Object.assign(sk.fromStatcastRow(r), { rawPa: Number(r.sample_pa || 0) }));
}
}
}
const led = await pageSafe(sb, 'ledger_entries',
'id, player_key, player_name, stat, line, side, outcome, game_date, p_win, quarantine_reason',
(q) => q.eq('sport', 'mlb').is('user_id', null)
.in('stat', ['hits', 'total_bases'])
.in('outcome', ['hit', 'miss']).not('p_win', 'is', null));
const clean = led.filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book')
&& String(r.game_date) > freezeDate);
const dates = [...new Set(clean.map((r) => r.game_date))].sort();
const starters = await opposingStarters(dates);
const players = new Map();
for (const r of clean) if (!players.has(r.player_key)) players.set(r.player_key, r.player_name);
const oppByPlayerDate = await opponentByPlayerDate(players);
const allowed = reg.candidateFeatures('mlb');
const rows = [];
for (const r of clean) {
const bat = batters.get(r.player_key);
if (!bat) continue;
const faced = oppByPlayerDate.get(`${r.player_key}|${r.game_date}`) || null;
const pit = faced ? pitchersById.get(Number(starters.get(`${r.game_date}|OPP:${faced}`))) || null : null;
const paRate = bat.rawPa > 0 ? Math.min(5.2, Math.max(2.0, bat.rawPa / GAMES_SO_FAR)) : null;
const under = String(r.side).toLowerCase() === 'under';
const won = r.outcome === 'hit' ? 1 : 0;
const champ = Number(r.p_win);
// The value projection (hits only — TB is refused by design, see skillProjection).
const proj = r.stat === 'hits'
? sk.projectSkill({ batter: bat, pitcher: pit, park: 1, archetype: null,
statType: 'hits', line: Number(r.line), expectedPa: paRate, allowed })
: null;
rows.push({
stat: r.stat, won, champ,
skill: proj ? (under ? 1 - proj.p_over_line : proj.p_over_line) : null,
residual: won - champ,
had_pitcher: !!pit,
// Candidate skill features, archetype-relevant, in probability space.
batter_barrel_pct: knownRate(bat.barrel_pct),
batter_hard_hit_pct: knownRate(bat.hard_hit_pct),
batter_exit_velo: knownRate(bat.avg_exit_velo),
batter_launch_angle: knownRate(bat.avg_launch_angle),
batter_k_pct: knownRate(bat.k_pct),
batter_bb_pct: knownRate(bat.bb_pct),
pitcher_k_pct: pit ? knownRate(pit.k_pct) : null,
pitcher_hard_hit_allowed: pit ? knownRate(pit.hard_hit_pct) : null,
});
}
const FEATURES = ['batter_barrel_pct', 'batter_hard_hit_pct', 'batter_exit_velo',
'batter_launch_angle', 'batter_k_pct', 'batter_bb_pct',
'pitcher_k_pct', 'pitcher_hard_hit_allowed'];
const perStat = {};
for (const stat of ['hits', 'total_bases']) {
const rs = rows.filter((r) => r.stat === stat);
if (rs.length === 0) continue;
const tests = FEATURES.length; // the Bonferroni denominator for THIS sweep
const gate = {};
for (const f of FEATURES) {
const xs = rs.map((r) => r[f]);
gate[f] = {
// 1. does it relate to the outcome at all?
raw_vs_outcome: cv.validateFactor(xs, rs.map((r) => r.won), tests),
// 2. THE ONE THAT COUNTS — is any of it NEW, i.e. missed by the counter?
marginal_vs_counter_residual: cv.validateFactor(xs, rs.map((r) => r.residual), tests),
};
}
const passed = FEATURES.filter((f) => gate[f].marginal_vs_counter_residual.validated);
perStat[stat] = {
n: rs.length,
base_rate: r4(mean(rs.map((r) => r.won))),
bonferroni_tests: tests,
features_passing_gate_on_marginal: passed,
gate,
};
}
// HEAD-TO-HEAD — hits only (the value engine covers hits).
const h2h = rows.filter((r) => r.stat === 'hits' && r.skill != null);
const ys = h2h.map((r) => r.won);
const bs = bootstrapDiff(h2h, 'skill', 'champ');
console.log(JSON.stringify({
premise_correction: 'statModel.js and correlateValidator.js do not exist in this repo. The gate was implemented to the spec in src/services/python/blueprints/unconventional.py (VALIDATION_REQUIREMENTS); supplementSystems.test.js inlines its own validateFactor and imports no implementation.',
out_of_sample: `skill profiles frozen ${freezeDate}; only game_date > ${freezeDate} scored`,
gate_spec: cv.VALIDATION_REQUIREMENTS,
per_stat_gate: perStat,
head_to_head_hits: {
n: h2h.length,
pitcher_coverage: r4(mean(h2h.map((r) => (r.had_pitcher ? 1 : 0)))),
base_rate: r4(mean(ys)),
resolution: { value_engine: r4(corrOf(h2h.map((r) => r.skill), ys)), counter: r4(corrOf(h2h.map((r) => r.champ), ys)) },
brier: { value_engine: r4(brier(h2h.map((r) => r.skill), ys)), counter: r4(brier(h2h.map((r) => r.champ), ys)) },
delta: bs,
verdict: !bs ? 'N-BLOCKED'
: (bs.ci_excludes_zero && bs.point > 0) ? 'VALUE ENGINE BEATS THE COUNTER'
: (bs.ci_excludes_zero && bs.point < 0) ? 'LOSES to the counter — iterate, do not promote'
: 'INCONCLUSIVE — do not promote',
},
}, null, 2));
process.exit(0);
}
if (require.main === module) {
main().catch((e) => { console.error(e); process.exit(1); });
}
// Exported so the read-integrity harness measures THE REAL FUNCTION.
module.exports = { READS };