Files
vyndr/scripts/tb-solo-and-interactions.js
T
builtbykev f61ec6b391 Read integrity, as-of context, and the shadow matchup resolve (A1-A7)
Seven orders of measurement-first repair. The served grade does not move.

A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS:
410-617 of 2,490 duplicated with an equal number never returned, while
rows.length matched the server exactly. safePaginate orders on a real unique
key, verifies the tuple at runtime, and THROWS on a query error instead of
treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn
through that reader, and defense_by_direction's distinct-n was likely below
the gate floor all along.

A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from
pg_index (the context tables are dated-composite and had no single unique
column). The unordered helper is deleted, not parked.

A3 — ledgerService and retentionService defaulted the SAME env var to
DIFFERENT versions, so no ledger row ever carried the marker eligibility
requires. One source now. model_snapshots settlement moved onto the cron:
15,484 -> 28,894 settled, repaired-champion 0 -> 7,556.

A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no
row at-or-before the date means the factor does not apply, never the nearest
row. Live path unchanged, proven 400/400 on real rows.

A5 — factor_inputs freezes what the factor READ, never the multiplier, so an
audit can recompute and check. It also recorded the finding: the three hits
factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by
the resolver and written by nothing.

A6/A7 — matchupKeys resolves those keys from the posted lineup plus the
schedule's probable pitchers, and fires the factors into a SHADOW freeze:
248 fires on 308 props, 245 of which would move the grade. The served
forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test
that decides whether they ever go live.

Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay
withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity
harness 34/34.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-11 22:49:56 -04:00

457 lines
22 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env node
'use strict';
/**
* tb-solo-and-interactions — PROVE BOTH, with the solo pass as the control.
*
* A feature can carry signal alone, only in combination, or both. Testing only
* interactions misses solo-real features AND cannot tell whether an interaction
* ADDS anything or merely re-encodes its own parts. So the solo result is the
* baseline every interaction has to beat.
*
* ── HOW "ADDS OVER ITS PARTS" IS MEASURED ────────────────────────────────
* Not by comparing two correlations by eye. The interaction's incremental
* signal is the PARTIAL correlation of the interaction term with the counter's
* residual, CONTROLLING FOR both component features:
*
* resid_I = I − OLS(I ~ A, B)
* resid_Y = Y − OLS(Y ~ A, B)
* incremental r = corr(resid_I, resid_Y)
*
* If the interaction is just barrel-rate wearing a different hat, regressing out
* barrel rate removes it and the incremental r collapses to ~0. That is exactly
* the redundancy the order is guarding against, and it is the difference between
* PASSES-AND-ADDS and PASSES-BUT-REDUNDANT.
*
* ── WHY THE COUNTER'S RESIDUAL IS THE TARGET ─────────────────────────────
* Correlating with the raw outcome rewards a feature for knowing what the
* counter already knows. Only the part the counter MISSES is new information,
* and only new information can improve the product. Both are reported; the
* residual one is the one that decides.
*
* ── THEORY FIRST ─────────────────────────────────────────────────────────
* Every interaction below is declared with a MECHANISM before it is measured.
* No blind pairwise search — with 8 features there are 28 pairs, and at α=.05
* roughly one in twenty returns "significant" from noise alone.
*
* SUPABASE_URL=... node scripts/tb-solo-and-interactions.js
*/
require('dotenv').config();
const { createClient } = require('@supabase/supabase-js');
const cv = require('../src/services/model/correlateValidator');
const sk = require('../src/services/model/skillProjection');
const reg = require('../src/services/model/featureRegistry');
const mlb = require('../src/services/adapters/mlbStatsAdapter');
const { knownRate, knownNumber } = require('../src/utils/known');
const { paginate } = require('../src/utils/safePaginate');
const { uniqueKeyFor } = require('../src/utils/tableKeys');
const SB_URL = process.env.SUPABASE_URL;
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
const PAGE = 1000;
const GAMES_SO_FAR = Number(process.env.STAGEA_GAMES_SO_FAR || 103);
const r4 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 10000) / 10000);
const mean = (a) => (a.length ? a.reduce((x, y) => x + y, 0) / a.length : null);
const brier = (ps, ys) => (ps.length ? ps.reduce((s, p, i) => s + (p - ys[i]) ** 2, 0) / ps.length : null);
/** OLS residuals of y on the given predictor columns (with intercept). */
function olsResiduals(y, Xcols) {
const n = y.length;
const p = Xcols.length + 1;
const X = [];
for (let i = 0; i < n; i += 1) {
const row = [1];
for (const c of Xcols) row.push(c[i]);
X.push(row);
}
// Normal equations (X'X) b = X'y, solved by Gauss-Jordan. p is 2-4 here.
const XtX = Array.from({ length: p }, () => new Array(p).fill(0));
const Xty = new Array(p).fill(0);
for (let i = 0; i < n; i += 1) {
for (let a = 0; a < p; a += 1) {
Xty[a] += X[i][a] * y[i];
for (let b = 0; b < p; b += 1) XtX[a][b] += X[i][a] * X[i][b];
}
}
const M = XtX.map((row, i) => [...row, Xty[i]]);
for (let col = 0; col < p; col += 1) {
let piv = col;
for (let r = col + 1; r < p; r += 1) if (Math.abs(M[r][col]) > Math.abs(M[piv][col])) piv = r;
if (Math.abs(M[piv][col]) < 1e-12) return null; // singular → cannot control honestly
[M[col], M[piv]] = [M[piv], M[col]];
const d = M[col][col];
for (let k = col; k <= p; k += 1) M[col][k] /= d;
for (let r = 0; r < p; r += 1) {
if (r === col) continue;
const f = M[r][col];
for (let k = col; k <= p; k += 1) M[r][k] -= f * M[col][k];
}
}
const beta = M.map((row) => row[p]);
return y.map((v, i) => v - X[i].reduce((s, xv, j) => s + xv * beta[j], 0));
}
/**
* Partial correlation of a with b, controlling for the columns in ctrl.
*
* COLLINEARITY IS CHECKED FIRST, and this is not pedantry — it caught a real
* error in this very script. The archetype-power proxy was defined as
* `barrel_pct / LEAGUE.barrel_pct`, an exact linear function of barrel_pct, so
* "control for both components" was rank-deficient and the partial correlation
* it produced (-0.132, the only one that looked like an incremental finding) was
* an artifact of a singular design matrix. The Gauss-Jordan pivot test missed it
* because the two columns differ by a scale factor, which keeps the pivot well
* above an absolute epsilon. Scale-free pairwise correlation catches it.
*/
function partialCorr(a, b, ctrl) {
for (let i = 0; i < ctrl.length; i += 1) {
for (let j = i + 1; j < ctrl.length; j += 1) {
const rr = cv.pearson(ctrl[i], ctrl[j]).r;
if (rr !== null && Math.abs(rr) > 0.999) return null; // same variable twice
}
}
const ra = olsResiduals(a, ctrl);
const rb = olsResiduals(b, ctrl);
if (!ra || !rb) return null;
return cv.pearson(ra, rb).r;
}
/** Rows where every named key is known — the honest common sample. */
function completeRows(rows, keys) {
return rows.filter((r) => keys.every((k) => knownNumber(r[k]) !== null));
}
function makeRnd(seed) {
let s = seed >>> 0;
return () => { s ^= s << 13; s >>>= 0; s ^= s >>> 17; s ^= s << 5; s >>>= 0; return s / 4294967296; };
}
function bootstrapDiff(rows, keyA, keyB, iters = 4000, seed = 20260804) {
if (rows.length < 30) return null;
const rnd = makeRnd(seed);
const n = rows.length;
const diffs = [];
for (let it = 0; it < iters; it += 1) {
const ys = []; const a = []; const b = [];
for (let i = 0; i < n; i += 1) {
const r = rows[Math.floor(rnd() * n)];
ys.push(r.won); a.push(r[keyA]); b.push(r[keyB]);
}
const ca = cv.pearson(a, ys).r; const cb = cv.pearson(b, ys).r;
if (ca == null || cb == null) continue;
diffs.push(ca - cb);
}
if (diffs.length < 100) return null;
diffs.sort((x, y) => x - y);
const q = (pp) => r4(diffs[Math.floor(pp * (diffs.length - 1))]);
const ci = [q(0.025), q(0.975)];
return {
point: r4(cv.pearson(rows.map((r) => r[keyA]), rows.map((r) => r.won)).r
- cv.pearson(rows.map((r) => r[keyB]), rows.map((r) => r.won)).r),
ci95: ci, ci_excludes_zero: ci[0] > 0 || ci[1] < 0,
};
}
// ── FIX A2 (2026-08-09) — THE SAFE WALK ───────────────────────────────────
// `page()` above walks with no ORDER BY. Measured on production, that returned
// the correct row COUNT and the wrong ROWS: up to 33.6% of a read came back
// twice while an equal share never came back at all. `pageSafe` routes the same
// call through `src/utils/safePaginate`, which orders on a UNIQUE key on every
// page, verifies uniqueness at runtime, and THROWS on a query error instead of
// treating it as end-of-data.
//
// `page()` SURVIVES only for the context tables (statcast_aggregates,
// batter_spray, team_defense, platoon_splits, park_dimensions, game_context...).
// Those have COMPOSITE primary keys with no single unique column, so
// safePaginate cannot express them. They measure 0% corruption today; making
// them safe needs a composite-key ordering the helper does not yet have. Do not
// use `page()` for ledger_entries or model_snapshots.
async function pageSafe(sb, table, select, apply, key = uniqueKeyFor(table)) {
return paginate(() => apply(sb.from(table).select(select)),
{ key, pageSize: PAGE, label: `${table}` });
}
/** THE MEASURED READS — main() and the harness call the same functions. */
const READS = {
snaps: (sb) => pageSafe(sb, 'model_snapshots', 'id, player_key, game_date, archetype',
(q) => q.eq('sport', 'mlb').eq('stat', 'total_bases').not('archetype', 'is', null)),
ledger: (sb) => pageSafe(sb, 'ledger_entries',
'id, player_key, player_name, stat, line, side, outcome, game_date, p_win, quarantine_reason',
(q) => q.eq('sport', 'mlb').is('user_id', null).eq('stat', 'total_bases')
.in('outcome', ['hit', 'miss']).not('p_win', 'is', null)),
};
async function opposingStarters(dates) {
const m = new Map();
for (const d of dates) {
let games = [];
try { games = await mlb.getScheduleWithPitchers(d); } catch { games = []; }
for (const g of games) {
if (!g.home || !g.away) continue;
if (g.home.probablePitcher) m.set(`${d}|OPP:${g.home.team}`, g.home.probablePitcher.id);
if (g.away.probablePitcher) m.set(`${d}|OPP:${g.away.team}`, g.away.probablePitcher.id);
}
}
return m;
}
async function opponentByPlayerDate(players) {
const map = new Map();
for (const [key, name] of players) {
try {
const found = await mlb.searchPlayer(name);
if (!found || !found.id) continue;
const log = await mlb.getPlayerGameLog(found.id);
for (const g of log || []) {
if (g && g.date && g.opponent) map.set(`${key}|${String(g.date).slice(0, 10)}`, g.opponent);
}
} catch { /* no log → no pitcher */ }
}
return map;
}
const SOLO = ['batter_barrel_pct', 'batter_hard_hit_pct', 'batter_exit_velo',
'batter_launch_angle', 'batter_k_pct', 'batter_bb_pct',
'pitcher_k_pct', 'pitcher_hard_hit_allowed'];
/** STEP 2 — theory first. Every interaction declares its mechanism. */
const INTERACTIONS = [
{
key: 'launch_x_exit_velo',
components: ['batter_launch_angle', 'batter_exit_velo'],
mechanism: 'Extra bases need BOTH conditions: hit hard AND hit in the air. A 105-mph ground ball is an out; a 25-degree popup is an out. Neither factor alone predicts bases, which is precisely why each may fail solo and the product may not.',
build: (r) => r.batter_launch_angle * r.batter_exit_velo,
},
{
key: 'exitvelo_x_pitcher_suppression',
components: ['batter_exit_velo', 'pitcher_hard_hit_allowed'],
mechanism: 'A hitter only realises his contact quality against a pitcher who permits contact quality. Elite suppression should attenuate a power bat; a contact-permitting arm should amplify it. The effect is conditional by construction.',
build: (r) => r.batter_exit_velo * r.pitcher_hard_hit_allowed,
},
{
key: 'barrel_x_power_archetype',
components: ['batter_barrel_pct', 'archetype_power'],
mechanism: 'ARCHETYPE-CONDITIONAL. Barrels convert to extra bases for hitters whose lane is power; for a speed/contact profile the same barrel rate is a rarer event on a swing built for something else. This is Discipline 2 stated as a testable interaction. NOTE: it is currently UNTESTABLE — statcast rows carry no archetype label, and the barrel-relative proxy is an exact linear function of barrel_pct, so controlling for both components is rank-deficient. It needs a real archetype classification joined in.',
build: (r) => r.batter_barrel_pct * r.archetype_power,
},
{
key: 'batterK_x_pitcherK',
components: ['batter_k_pct', 'pitcher_k_pct'],
mechanism: 'Strikeout risk compounds multiplicatively (log5 is exactly this shape). A high-K bat against a high-K arm loses plate appearances to strikeouts, and a PA lost is a base opportunity that never happens — so it suppresses total bases through OPPORTUNITY, not contact quality.',
build: (r) => r.batter_k_pct * r.pitcher_k_pct,
},
];
/** Latest settled game date in the pull — used to detect that the profile
* freeze now sits AFTER the data, i.e. no clean out-of-sample window exists. */
function clean0Max(rows) {
return (rows || []).reduce((mx, r) => (String(r.game_date) > mx ? String(r.game_date) : mx), '');
}
async function main() {
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
const statcast = await pageSafe(sb, 'statcast_aggregates', '*', (q) => q.eq('sport', 'mlb'));
const freezeDate = statcast.reduce((mx, r) => (String(r.updated_at) > mx ? String(r.updated_at) : mx), '').slice(0, 10);
const batters = new Map(); const pitchersById = new Map();
for (const r of statcast) {
if (r.role === 'pitcher' && r.source_id != null) pitchersById.set(Number(r.source_id), sk.fromStatcastRow(r));
if (r.player_key && r.role === 'batter') {
const prev = batters.get(r.player_key);
if (!prev || Number(r.sample_pa || 0) > Number(prev.rawPa || 0)) {
batters.set(r.player_key, Object.assign(sk.fromStatcastRow(r), { rawPa: Number(r.sample_pa || 0) }));
}
}
}
// REAL ARCHETYPE LABELS. The barrel-relative proxy was a clipped monotone
// transform of barrel_pct, so `barrel x proxy` measured NONLINEARITY IN BARREL,
// not an archetype interaction — it could never have tested Discipline 2.
// model_snapshots carries the actual classification per prop, so the
// conditioning variable is now a genuine BOMBER indicator, which is
// categorical and therefore not a transform of barrel at all.
const snaps = await READS.snaps(sb);
const archetypeBy = new Map();
for (const r of snaps) if (r.player_key && r.game_date) archetypeBy.set(`${r.player_key}|${r.game_date}`, r.archetype);
const led = await READS.ledger(sb);
// POINT-IN-TIME IS NO LONGER AVAILABLE FROM THIS TABLE.
//
// `statcast_aggregates` is upserted in place and keeps one as-of date. The
// first skill backtest was honest only by accident: the nightly refresh was
// unreachable code, so the table sat frozen at 2026-07-21 — BEFORE the settled
// window. Repairing that cron (correct for production) refreshed it to today,
// and every prior version is gone.
//
// So scoring a 2026-07-25 game now uses a season aggregate that CONTAINS that
// game. `statcast_history` (added this session) fixes it going forward; it has
// one day of data, which is not yet a window. Until it fills, results here are
// DIRECTIONAL AND CONTAMINATED, labelled as such, and are NOT gate verdicts.
const contaminated = String(freezeDate) >= String(clean0Max(led));
const clean = led.filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book')
&& (contaminated ? true : String(r.game_date) > freezeDate));
const dates = [...new Set(clean.map((r) => r.game_date))].sort();
const starters = await opposingStarters(dates);
const players = new Map();
for (const r of clean) if (!players.has(r.player_key)) players.set(r.player_key, r.player_name);
const oppByPlayerDate = await opponentByPlayerDate(players);
if (process.env.TB_DEBUG === '1') {
console.error(`[debug] statcast rows=${statcast.length} freeze=${freezeDate} batters=${batters.size} pitchers=${pitchersById.size}`);
console.error(`[debug] ledger tb rows=${led.length} clean(after freeze)=${clean.length}`);
const sampleKeys = clean.slice(0, 5).map((r) => r.player_key);
console.error(`[debug] sample ledger player_keys=${JSON.stringify(sampleKeys)}`);
console.error(`[debug] sample statcast keys=${JSON.stringify([...batters.keys()].slice(0, 5))}`);
console.error(`[debug] matches in sample=${sampleKeys.filter((k) => batters.has(k)).length}/5`);
}
const allowed = reg.candidateFeaturesForStat('mlb', 'total_bases');
const rows = [];
for (const r of clean) {
const bat = batters.get(r.player_key);
if (!bat) continue;
const faced = oppByPlayerDate.get(`${r.player_key}|${r.game_date}`) || null;
const pit = faced ? pitchersById.get(Number(starters.get(`${r.game_date}|OPP:${faced}`))) || null : null;
const paRate = bat.rawPa > 0 ? Math.min(5.2, Math.max(2.0, bat.rawPa / GAMES_SO_FAR)) : null;
const under = String(r.side).toLowerCase() === 'under';
const won = r.outcome === 'hit' ? 1 : 0;
const champ = Number(r.p_win);
const proj = sk.projectSkill({
batter: bat, pitcher: pit, park: 1, archetype: null,
statType: 'total_bases', line: Number(r.line), expectedPa: paRate, allowed,
});
// The REAL archetype for this prop — a 0/1 power indicator, categorical and
// independent of barrel_pct by construction.
const arch = archetypeBy.get(`${r.player_key}|${r.game_date}`) || null;
const archetypePower = arch == null ? null : (String(arch).toUpperCase() === 'BOMBER' ? 1 : 0);
rows.push({
won, champ, residual: won - champ,
skill: proj ? (under ? 1 - proj.p_over_line : proj.p_over_line) : null,
had_pitcher: !!pit,
batter_barrel_pct: knownRate(bat.barrel_pct),
batter_hard_hit_pct: knownRate(bat.hard_hit_pct),
batter_exit_velo: knownRate(bat.avg_exit_velo),
batter_launch_angle: knownRate(bat.avg_launch_angle),
batter_k_pct: knownRate(bat.k_pct),
batter_bb_pct: knownRate(bat.bb_pct),
pitcher_k_pct: pit ? knownRate(pit.k_pct) : null,
pitcher_hard_hit_allowed: pit ? knownRate(pit.hard_hit_pct) : null,
archetype_power: archetypePower,
archetype: arch,
});
}
// Bonferroni denominator = every test in this family (solo + interaction).
// ── CUMULATIVE BONFERRONI ─────────────────────────────────────────────
// The denominator is every DISTINCT hypothesis this programme has tested, not
// this run's. A per-session count gives each new order a fresh, generous alpha
// and lets the false-positive rate compound silently.
const tl = require('../src/services/model/testLedger');
const mcStore = tl.supabaseStore(sb);
const mc = await tl.recordAndCount(mcStore, [
...SOLO.map((f) => ({ sport: 'mlb', stat: 'total_bases', archetype: null, interaction: `solo:${f}`, target: 'counter_residual' })),
...INTERACTIONS.map((x) => ({ sport: 'mlb', stat: 'total_bases', archetype: null, interaction: x.key, target: 'counter_residual' })),
]);
const TESTS = mc.cumulative_tests;
// ── STEP 1 — SOLO PASS (the control) ────────────────────────────────────
const solo = {};
for (const f of SOLO) {
const rs = completeRows(rows, [f]);
solo[f] = {
n: rs.length,
vs_outcome: cv.validateFactor(rs.map((r) => r[f]), rs.map((r) => r.won), TESTS),
vs_counter_residual: cv.validateFactor(rs.map((r) => r[f]), rs.map((r) => r.residual), TESTS),
};
}
// ── STEP 3 — INTERACTIONS, each against its own solo baseline ───────────
const interactions = {};
for (const ix of INTERACTIONS) {
const keys = [...ix.components];
const rs = completeRows(rows, keys);
if (rs.length < 30) { interactions[ix.key] = { mechanism: ix.mechanism, n: rs.length, verdict: 'UNTESTABLE — no common sample' }; continue; }
const I = rs.map(ix.build);
const Y = rs.map((r) => r.residual);
const ctrl = keys.map((k) => rs.map((r) => r[k]));
const gate = cv.validateFactor(I, Y, TESTS);
const incremental = partialCorr(I, Y, ctrl);
// The best solo |r| among its own components, on the SAME rows.
const componentSolo = keys.map((k) => ({
feature: k, r: r4(cv.pearson(rs.map((r) => r[k]), Y).r),
}));
const bestComponent = Math.max(...componentSolo.map((c) => Math.abs(c.r ?? 0)));
let verdict;
if (incremental === null) verdict = 'UNTESTABLE — controls are collinear';
else if (gate.validated && Math.abs(incremental) >= cv.VALIDATION_REQUIREMENTS.min_pearson_r) verdict = 'PASSES-AND-ADDS';
else if (gate.validated) verdict = 'PASSES-BUT-REDUNDANT';
else if (rs.length < cv.VALIDATION_REQUIREMENTS.min_historical_instances) verdict = 'UNDERPOWERED — n below the gate';
else verdict = 'FAILS';
interactions[ix.key] = {
mechanism: ix.mechanism,
components: keys,
n: rs.length,
raw_r_vs_residual: gate.pearson_r,
gate: { validated: gate.validated, reason: gate.reason, p_value: gate.p_value, corrected_alpha: gate.corrected_alpha, underpowered: !!gate.underpowered },
component_solo_r_same_rows: componentSolo,
best_component_abs_r: r4(bestComponent),
INCREMENTAL_partial_r: r4(incremental),
adds_over_components: incremental !== null && Math.abs(incremental) > bestComponent,
verdict,
};
}
// ── STEP 4 — COMBINED vs COUNTER (valid at this n; the gate is not) ─────
const h2h = rows.filter((r) => r.skill != null);
const ys = h2h.map((r) => r.won);
const bs = bootstrapDiff(h2h, 'skill', 'champ');
console.log(JSON.stringify({
stat: 'total_bases',
VALIDITY: contaminated
? 'CONTAMINATED / DIRECTIONAL ONLY — statcast_aggregates now carries a single as-of date (' + freezeDate + ') that is AFTER the settled games, so season profiles contain the games being predicted. These are NOT gate verdicts. statcast_history (new) makes point-in-time possible from tomorrow.'
: `CLEAN out-of-sample: profiles frozen ${freezeDate}; only game_date > ${freezeDate} scored`,
contaminated,
rows_scored: rows.length,
gate_spec: cv.VALIDATION_REQUIREMENTS,
bonferroni_tests: TESTS,
multiple_comparisons: { ...mc, note: 'cumulative across the programme lifetime, not this session' },
n_gap_note: `the gate needs ${cv.VALIDATION_REQUIREMENTS.min_historical_instances} rows; this run has ${rows.length}`,
archetype_coverage: {
labelled: rows.filter((r) => r.archetype).length,
bomber: rows.filter((r) => r.archetype_power === 1).length,
other: rows.filter((r) => r.archetype_power === 0).length,
},
step1_solo_baseline: solo,
step3_interactions: interactions,
step4_combined_vs_counter: {
n: h2h.length,
pitcher_coverage: r4(mean(h2h.map((r) => (r.had_pitcher ? 1 : 0)))),
base_rate: r4(mean(ys)),
resolution: { skill_tb: r4(cv.pearson(h2h.map((r) => r.skill), ys).r), counter: r4(cv.pearson(h2h.map((r) => r.champ), ys).r) },
brier: { skill_tb: r4(brier(h2h.map((r) => r.skill), ys)), counter: r4(brier(h2h.map((r) => r.champ), ys)) },
delta: bs,
verdict: !bs ? 'N-BLOCKED'
: (bs.ci_excludes_zero && bs.point > 0) ? 'SKILL TB BEATS THE COUNTER'
: (bs.ci_excludes_zero && bs.point < 0) ? 'LOSES to the counter'
: 'INCONCLUSIVE',
},
}, null, 2));
process.exit(0);
}
if (require.main === module) {
main().catch((e) => { console.error(e); process.exit(1); });
}
// Exported so the read-integrity harness measures THE REAL FUNCTION.
module.exports = { READS };