Files
vyndr/scripts/pitcher-prove-k.js
T
builtbykev f61ec6b391 Read integrity, as-of context, and the shadow matchup resolve (A1-A7)
Seven orders of measurement-first repair. The served grade does not move.

A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS:
410-617 of 2,490 duplicated with an equal number never returned, while
rows.length matched the server exactly. safePaginate orders on a real unique
key, verifies the tuple at runtime, and THROWS on a query error instead of
treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn
through that reader, and defense_by_direction's distinct-n was likely below
the gate floor all along.

A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from
pg_index (the context tables are dated-composite and had no single unique
column). The unordered helper is deleted, not parked.

A3 — ledgerService and retentionService defaulted the SAME env var to
DIFFERENT versions, so no ledger row ever carried the marker eligibility
requires. One source now. model_snapshots settlement moved onto the cron:
15,484 -> 28,894 settled, repaired-champion 0 -> 7,556.

A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no
row at-or-before the date means the factor does not apply, never the nearest
row. Live path unchanged, proven 400/400 on real rows.

A5 — factor_inputs freezes what the factor READ, never the multiplier, so an
audit can recompute and check. It also recorded the finding: the three hits
factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by
the resolver and written by nothing.

A6/A7 — matchupKeys resolves those keys from the posted lineup plus the
schedule's probable pitchers, and fires the factors into a SHADOW freeze:
248 fires on 308 props, 245 of which would move the grade. The served
forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test
that decides whether they ever go live.

Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay
withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity
harness 34/34.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-11 22:49:56 -04:00

381 lines
20 KiB
JavaScript

#!/usr/bin/env node
'use strict';
/**
* pitcher-prove-k — STRIKEOUTS through the both-ways gate.
*
* Same bar as everything else: solo pass as the control, theory-first
* interactions each measured against their own components, gate at n>=500 /
* |r|>=0.15 / p<0.05 / Bonferroni, then head-to-head vs the counter.
*
* THE THEORIZED SIGNAL-CARRIER is `stuff x opposing-lineup K-rate`. An elite
* strikeout arm against a contact lineup that never whiffs is a different bet
* from the same arm against a three-true-outcomes lineup, and neither side says
* it alone — the pitcher analogue of the batter model's contact-quality term.
* The lineup rate is built from the OPPOSING TEAM'S OWN BATTERS (roster join to
* their statcast K%), not from a league constant, or the interaction would be a
* relabelled copy of the pitcher's own rate.
*
* VALIDITY: statcast_aggregates still carries one as-of date (2026-08-03) and
* `statcast_history` has one day, so there is no point-in-time window yet.
* Results here are CONTAMINATED / DIRECTIONAL and are not gate verdicts.
*
* SUPABASE_URL=... node scripts/pitcher-prove-k.js
*/
require('dotenv').config();
const { createClient } = require('@supabase/supabase-js');
const cv = require('../src/services/model/correlateValidator');
const pe = require('../src/services/model/pitcherEngine');
const sk = require('../src/services/model/skillProjection');
const mlb = require('../src/services/adapters/mlbStatsAdapter');
const { knownRate, knownNumber } = require('../src/utils/known');
const { paginate } = require('../src/utils/safePaginate');
const { uniqueKeyFor } = require('../src/utils/tableKeys');
const SB_URL = process.env.SUPABASE_URL;
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
const PAGE = 1000;
const r4 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 10000) / 10000);
const mean = (a) => (a.length ? a.reduce((x, y) => x + y, 0) / a.length : null);
const brier = (ps, ys) => (ps.length ? ps.reduce((s, p, i) => s + (p - ys[i]) ** 2, 0) / ps.length : null);
function olsResiduals(y, Xcols) {
const n = y.length; const p = Xcols.length + 1;
const X = []; for (let i = 0; i < n; i += 1) { const row = [1]; for (const c of Xcols) row.push(c[i]); X.push(row); }
const XtX = Array.from({ length: p }, () => new Array(p).fill(0)); const Xty = new Array(p).fill(0);
for (let i = 0; i < n; i += 1) for (let a = 0; a < p; a += 1) {
Xty[a] += X[i][a] * y[i];
for (let b = 0; b < p; b += 1) XtX[a][b] += X[i][a] * X[i][b];
}
const M = XtX.map((row, i) => [...row, Xty[i]]);
for (let col = 0; col < p; col += 1) {
let piv = col;
for (let r = col + 1; r < p; r += 1) if (Math.abs(M[r][col]) > Math.abs(M[piv][col])) piv = r;
if (Math.abs(M[piv][col]) < 1e-12) return null;
[M[col], M[piv]] = [M[piv], M[col]];
const d = M[col][col];
for (let k = col; k <= p; k += 1) M[col][k] /= d;
for (let r = 0; r < p; r += 1) { if (r === col) continue; const f = M[r][col]; for (let k = col; k <= p; k += 1) M[r][k] -= f * M[col][k]; }
}
const beta = M.map((row) => row[p]);
return y.map((v, i) => v - X[i].reduce((s, xv, j) => s + xv * beta[j], 0));
}
function partialCorr(a, b, ctrl) {
for (let i = 0; i < ctrl.length; i += 1) for (let j = i + 1; j < ctrl.length; j += 1) {
const rr = cv.pearson(ctrl[i], ctrl[j]).r;
if (rr !== null && Math.abs(rr) > 0.999) return null; // same variable twice
}
const ra = olsResiduals(a, ctrl); const rb = olsResiduals(b, ctrl);
if (!ra || !rb) return null;
return cv.pearson(ra, rb).r;
}
function makeRnd(seed) { let s = seed >>> 0; return () => { s ^= s << 13; s >>>= 0; s ^= s >>> 17; s ^= s << 5; s >>>= 0; return s / 4294967296; }; }
function bootstrapDiff(rows, kA, kB, iters = 4000, seed = 20260805) {
if (rows.length < 30) return null;
const rnd = makeRnd(seed); const n = rows.length; const diffs = [];
for (let it = 0; it < iters; it += 1) {
const ys = []; const a = []; const b = [];
for (let i = 0; i < n; i += 1) { const r = rows[Math.floor(rnd() * n)]; ys.push(r.won); a.push(r[kA]); b.push(r[kB]); }
const ca = cv.pearson(a, ys).r; const cb = cv.pearson(b, ys).r;
if (ca == null || cb == null) continue;
diffs.push(ca - cb);
}
if (diffs.length < 100) return null;
diffs.sort((x, y) => x - y);
const q = (pp) => r4(diffs[Math.floor(pp * (diffs.length - 1))]);
const ci = [q(0.025), q(0.975)];
return { point: r4(cv.pearson(rows.map((r) => r[kA]), rows.map((r) => r.won)).r - cv.pearson(rows.map((r) => r[kB]), rows.map((r) => r.won)).r), ci95: ci, ci_excludes_zero: ci[0] > 0 || ci[1] < 0 };
}
// ── THE SAFE WALK (Fix A2/A2b) ────────────────────────────────────────────
// This script used to walk pages with `.range()` and NO ORDER BY. Measured on
// production, that returned the correct row COUNT and the wrong ROWS: up to
// 33.6% of a read came back twice while an equal share never came back at all,
// so `rows.length` looked perfect while a fifth of the sample was missing.
//
// `pageSafe` routes every read through `src/utils/safePaginate`: a stable ORDER
// BY on the table's real UNIQUE key — single OR composite, looked up from
// `src/utils/tableKeys` rather than assumed — a runtime tuple-uniqueness check,
// and a THROWN error instead of a silent stop. The old unordered helper is gone
// rather than left beside it, because a dead broken helper is an invitation.
async function pageSafe(sb, table, select, apply, key = uniqueKeyFor(table)) {
return paginate(() => apply(sb.from(table).select(select)),
{ key, pageSize: PAGE, label: `${table}` });
}
/** THE MEASURED READS — main() and the harness call the same functions. */
const READS = {
ledger: (sb) => pageSafe(sb, 'ledger_entries',
'id, player_key, player_name, line, side, outcome, game_date, p_win, quarantine_reason, game_id',
(q) => q.eq('sport', 'mlb').is('user_id', null).eq('stat', 'strikeouts')
.in('outcome', ['hit', 'miss']).not('p_win', 'is', null)),
};
const SOLO = ['pitcher_whiff_pct', 'pitcher_k_pct', 'pitcher_chase_pct', 'pitcher_gb_pct',
'pitcher_arm_angle', 'opposing_lineup_k_rate'];
const INTERACTIONS = [
{
key: 'stuff_x_lineup_k_rate',
components: ['pitcher_whiff_pct', 'opposing_lineup_k_rate'],
mechanism: 'THE theorized carrier. Strikeouts need a pitcher who can miss bats AND a lineup that can be missed. An elite arm against a contact lineup and a modest arm against a whiff-prone one can produce the same count, so neither factor alone orders the props — the product should.',
build: (r) => r.pitcher_whiff_pct * r.opposing_lineup_k_rate,
},
{
key: 'stuff_x_power_archetype',
components: ['pitcher_whiff_pct', 'archetype_flame'],
mechanism: 'ARCHETYPE-CONDITIONAL. Stuff should govern strikeouts more for a power arm than for a finesse arm, whose Ks come from chase and sequencing. Discipline 2 as a testable claim, with a categorical conditioner independent of whiff by construction.',
build: (r) => r.pitcher_whiff_pct * r.archetype_flame,
},
{
key: 'chase_x_lineup_k_rate',
components: ['pitcher_chase_pct', 'opposing_lineup_k_rate'],
mechanism: 'The finesse channel: expanding the zone only works against a lineup that will chase. Same shape as the stuff term, different mechanism, so it is tested separately rather than assumed to be the same effect.',
build: (r) => r.pitcher_chase_pct * r.opposing_lineup_k_rate,
},
];
async function main() {
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
const statcast = await pageSafe(sb, 'statcast_aggregates', '*', (q) => q.eq('sport', 'mlb'));
const freezeDate = statcast.reduce((mx, r) => (String(r.updated_at) > mx ? String(r.updated_at) : mx), '').slice(0, 10);
const pitchByKey = new Map(); const batterByKey = new Map();
for (const r of statcast) {
const prof = sk.fromStatcastRow(r);
if (r.role === 'pitcher' && r.player_key) pitchByKey.set(r.player_key, prof);
if (r.role === 'batter' && r.player_key) batterByKey.set(r.player_key, prof);
}
const led = await pageSafe(sb, 'ledger_entries',
'id, player_key, player_name, stat, line, side, outcome, game_date, p_win, quarantine_reason',
(q) => q.eq('sport', 'mlb').is('user_id', null).eq('stat', 'strikeouts')
.in('outcome', ['hit', 'miss']).not('p_win', 'is', null));
const clean = led.filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book'));
// ── OPPOSING LINEUP K-RATE, from the opposing team's OWN batters ────────
// Roster join, not a league constant: a constant would make the interaction a
// rescaled copy of the pitcher's own rate and guarantee a false "redundant".
const { nameKey } = require('../src/utils/playerName');
const teamKRate = new Map();
async function lineupKFor(teamName) {
if (!teamName) return null;
if (teamKRate.has(teamName)) return teamKRate.get(teamName);
let val = null;
try {
// The game log gives a full team NAME ("Cincinnati Reds"); resolveTeam
// wants an ABBREVIATION. Passing the name straight through silently
// resolved nothing and produced 0% lineup coverage on the first run — the
// theorized signal-carrier was not failing, it was never being tested.
const { NAME_TO_ABBR } = require('../src/services/environmentContext');
const abbr = /^[A-Z]{2,3}$/.test(String(teamName).trim())
? String(teamName).trim().toUpperCase()
: NAME_TO_ABBR[String(teamName).toLowerCase()];
if (!abbr) { teamKRate.set(teamName, null); return null; }
const team = await mlb.resolveTeam(abbr);
const roster = team && team.id ? await mlb.getTeamRoster(team.id) : null;
// PA-WEIGHTED, not a flat roster average. An unweighted mean counts a
// 12-PA September call-up the same as an everyday starter, which is not
// the lineup a pitcher faces. Weighting by each batter's own sample_pa is
// the closest honest approximation of "who actually bats" from data we
// already hold — and it needs no new sourcing at all.
let wSum = 0; let wK = 0; let counted = 0;
for (const p of roster || []) {
const prof = batterByKey.get(nameKey(p.name || p.fullName || ''));
if (!prof) continue;
const k = knownRate(prof.k_pct);
const pa = knownRate(prof.sample_pa);
if (k === null) continue;
const w = pa === null ? 0 : pa; // no PA read -> contributes nothing
if (w <= 0) continue;
wSum += w; wK += w * k; counted += 1;
}
if (counted >= 5 && wSum > 0) val = wK / wSum;
} catch { val = null; }
teamKRate.set(teamName, val);
return val;
}
// The opponent a pitcher faced on a date, from his own game log.
const oppBy = new Map();
const names = new Map();
for (const r of clean) if (!names.has(r.player_key)) names.set(r.player_key, r.player_name);
for (const [key, name] of names) {
try {
const found = await mlb.searchPlayer(name);
if (!found || !found.id) continue;
const log = await mlb.getPlayerGameLog(found.id, undefined, 'pitching');
for (const g of log || []) if (g && g.date && g.opponent) oppBy.set(`${key}|${String(g.date).slice(0, 10)}`, g.opponent);
} catch { /* no log → no lineup term */ }
}
const rows = [];
for (const r of clean) {
const prof = pitchByKey.get(r.player_key);
if (!prof) continue;
const opp = oppBy.get(`${r.player_key}|${r.game_date}`) || null;
const lineupK = opp ? await lineupKFor(opp) : null;
const cls = pe.classifyPitcher(prof);
const arch = cls ? cls.primary : null;
const under = String(r.side).toLowerCase() === 'under';
const won = r.outcome === 'hit' ? 1 : 0;
const champ = Number(r.p_win);
const proj = pe.projectStrikeouts({
pitcher: prof, lineupKRate: lineupK, archetype: arch,
role: 'starter', line: Number(r.line),
});
// The SAME model with the lineup term switched off, so the term's
// contribution is isolated rather than inferred.
const projNo = pe.projectStrikeouts({
pitcher: prof, lineupKRate: null, archetype: arch,
role: 'starter', line: Number(r.line),
});
rows.push({
won, champ, residual: won - champ,
pitch: proj ? (under ? 1 - proj.p_over_line : proj.p_over_line) : null,
pitch_nolineup: projNo ? (under ? 1 - projNo.p_over_line : projNo.p_over_line) : null,
lineup_applied: !!lineupK,
archetype: arch,
archetype_flame: arch == null ? null : (arch === 'FLAME' ? 1 : 0),
pitcher_whiff_pct: knownRate(prof.whiff_pct),
pitcher_k_pct: knownRate(prof.k_pct),
pitcher_chase_pct: knownRate(prof.chase_pct),
pitcher_gb_pct: knownRate(prof.gb_pct),
pitcher_arm_angle: knownRate(prof.arm_angle),
opposing_lineup_k_rate: lineupK,
});
}
// ── CUMULATIVE BONFERRONI ─────────────────────────────────────────────
// The denominator is every DISTINCT hypothesis this programme has tested, not
// this run's. A per-session count gives each new order a fresh, generous alpha
// and lets the false-positive rate compound silently.
const tl = require('../src/services/model/testLedger');
const mcStore = tl.supabaseStore(sb);
const mc = await tl.recordAndCount(mcStore, [
...SOLO.map((f) => ({ sport: 'mlb', stat: 'strikeouts', archetype: null, interaction: `solo:${f}`, target: 'counter_residual' })),
...INTERACTIONS.map((x) => ({ sport: 'mlb', stat: 'strikeouts', archetype: null, interaction: x.key, target: 'counter_residual' })),
]);
const TESTS = mc.cumulative_tests;
const complete = (keys) => rows.filter((r) => keys.every((k) => knownNumber(r[k]) !== null));
const solo = {};
for (const f of SOLO) {
const rs = complete([f]);
solo[f] = {
n: rs.length,
vs_outcome: cv.validateFactor(rs.map((r) => r[f]), rs.map((r) => r.won), TESTS),
vs_counter_residual: cv.validateFactor(rs.map((r) => r[f]), rs.map((r) => r.residual), TESTS),
};
}
const interactions = {};
for (const ix of INTERACTIONS) {
const rs = complete(ix.components);
if (rs.length < 20) { interactions[ix.key] = { mechanism: ix.mechanism, n: rs.length, verdict: 'UNTESTABLE — no common sample' }; continue; }
const I = rs.map(ix.build); const Y = rs.map((r) => r.residual);
const ctrl = ix.components.map((k) => rs.map((r) => r[k]));
const gate = cv.validateFactor(I, Y, TESTS);
const incr = partialCorr(I, Y, ctrl);
const parts = ix.components.map((k) => ({ feature: k, r: r4(cv.pearson(rs.map((r) => r[k]), Y).r) }));
const best = Math.max(...parts.map((p) => Math.abs(p.r ?? 0)));
interactions[ix.key] = {
mechanism: ix.mechanism, components: ix.components, n: rs.length,
raw_r_vs_residual: gate.pearson_r,
gate: { validated: gate.validated, reason: gate.reason, underpowered: !!gate.underpowered, rows_needed: gate.rows_needed ?? null },
component_solo_r: parts, best_component_abs_r: r4(best),
INCREMENTAL_partial_r: r4(incr),
adds_over_components: incr !== null && Math.abs(incr) > best,
verdict: incr === null ? 'UNTESTABLE — collinear controls'
: (gate.validated && Math.abs(incr) >= 0.15) ? 'PASSES-AND-ADDS'
: gate.validated ? 'PASSES-BUT-REDUNDANT'
: rs.length < 500 ? 'UNDERPOWERED — n below the gate' : 'FAILS',
};
}
// ── WITHIN-ARCHETYPE: the order's sharper hypothesis ────────────────────
// The interaction should matter MORE for finesse arms (SCALPEL/SINKER), whose
// strikeouts need a lineup that will chase or can be beaten, than for power
// arms (FLAME) whose stuff whiffs regardless of who is standing there. Pooling
// the two would average a real conditional effect toward zero — which is
// exactly the failure mode "test within archetype" exists to prevent.
const strata = {};
for (const [label, pred] of [
['FLAME (power)', (r) => r.archetype === 'FLAME'],
['non-FLAME (finesse/contact)', (r) => r.archetype && r.archetype !== 'FLAME'],
]) {
const rs = rows.filter((r) => pred(r)
&& knownNumber(r.pitcher_whiff_pct) !== null
&& knownNumber(r.opposing_lineup_k_rate) !== null);
if (rs.length < 15) { strata[label] = { n: rs.length, verdict: 'UNTESTABLE — stratum too thin' }; continue; }
const I = rs.map((r) => r.pitcher_whiff_pct * r.opposing_lineup_k_rate);
const Y = rs.map((r) => r.residual);
const ctrl = [rs.map((r) => r.pitcher_whiff_pct), rs.map((r) => r.opposing_lineup_k_rate)];
const incr = partialCorr(I, Y, ctrl);
const parts = [
{ feature: 'pitcher_whiff_pct', r: r4(cv.pearson(rs.map((r) => r.pitcher_whiff_pct), Y).r) },
{ feature: 'opposing_lineup_k_rate', r: r4(cv.pearson(rs.map((r) => r.opposing_lineup_k_rate), Y).r) },
];
const best = Math.max(...parts.map((p) => Math.abs(p.r ?? 0)));
strata[label] = {
n: rs.length,
raw_r_vs_residual: r4(cv.pearson(I, Y).r),
lineup_solo_r: parts[1].r,
component_solo_r: parts,
best_component_abs_r: r4(best),
INCREMENTAL_partial_r: r4(incr),
adds_over_components: incr !== null && Math.abs(incr) > best,
gate: cv.validateFactor(I, Y, TESTS),
verdict: incr === null ? 'UNTESTABLE — collinear'
: rs.length < 500 ? 'UNDERPOWERED — n below the gate'
: (Math.abs(incr) >= 0.15 ? 'ADDS' : 'REDUNDANT'),
};
}
const h2h = rows.filter((r) => r.pitch != null);
const ys = h2h.map((r) => r.won);
const bs = bootstrapDiff(h2h, 'pitch', 'champ');
const bsNoLineup = bootstrapDiff(h2h.filter((r) => r.pitch_nolineup != null), 'pitch_nolineup', 'champ');
console.log(JSON.stringify({
stat: 'strikeouts',
VALIDITY: `CONTAMINATED / DIRECTIONAL — statcast carries one as-of date (${freezeDate}); statcast_history has no window yet. NOT gate verdicts.`,
rows_scored: rows.length,
lineup_coverage: r4(mean(rows.map((r) => (r.lineup_applied ? 1 : 0)))),
archetype_mix: rows.reduce((a, r) => { const k = r.archetype || 'unclassified'; a[k] = (a[k] || 0) + 1; return a; }, {}),
gate_spec: cv.VALIDATION_REQUIREMENTS,
bonferroni_tests: TESTS,
multiple_comparisons: { ...mc, note: 'cumulative across the programme lifetime, not this session' },
step1_solo_baseline: solo,
step3_interactions: interactions,
step3b_within_archetype_carrier: strata,
step4_vs_counter: {
n: h2h.length,
base_rate: r4(mean(ys)),
resolution: { pitch_v1: r4(cv.pearson(h2h.map((r) => r.pitch), ys).r), counter: r4(cv.pearson(h2h.map((r) => r.champ), ys).r) },
brier: { pitch_v1: r4(brier(h2h.map((r) => r.pitch), ys)), counter: r4(brier(h2h.map((r) => r.champ), ys)) },
delta: bs,
// Isolating the lineup term: does including it help or hurt?
without_lineup_term: {
resolution: r4(cv.pearson(h2h.filter((r) => r.pitch_nolineup != null).map((r) => r.pitch_nolineup),
h2h.filter((r) => r.pitch_nolineup != null).map((r) => r.won)).r),
delta_vs_counter: bsNoLineup,
},
verdict: !bs ? 'N-BLOCKED — too few rows to bootstrap'
: (bs.ci_excludes_zero && bs.point > 0) ? 'BEATS THE COUNTER'
: (bs.ci_excludes_zero && bs.point < 0) ? 'LOSES to the counter' : 'INCONCLUSIVE',
},
}, null, 2));
process.exit(0);
}
if (require.main === module) {
main().catch((e) => { console.error(e); process.exit(1); });
}
// Exported so the read-integrity harness measures THE REAL FUNCTION.
module.exports = { READS };