Files
vyndr/scripts/proven-status.js
T
builtbykev f61ec6b391 Read integrity, as-of context, and the shadow matchup resolve (A1-A7)
Seven orders of measurement-first repair. The served grade does not move.

A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS:
410-617 of 2,490 duplicated with an equal number never returned, while
rows.length matched the server exactly. safePaginate orders on a real unique
key, verifies the tuple at runtime, and THROWS on a query error instead of
treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn
through that reader, and defense_by_direction's distinct-n was likely below
the gate floor all along.

A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from
pg_index (the context tables are dated-composite and had no single unique
column). The unordered helper is deleted, not parked.

A3 — ledgerService and retentionService defaulted the SAME env var to
DIFFERENT versions, so no ledger row ever carried the marker eligibility
requires. One source now. model_snapshots settlement moved onto the cron:
15,484 -> 28,894 settled, repaired-champion 0 -> 7,556.

A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no
row at-or-before the date means the factor does not apply, never the nearest
row. Live path unchanged, proven 400/400 on real rows.

A5 — factor_inputs freezes what the factor READ, never the multiplier, so an
audit can recompute and check. It also recorded the finding: the three hits
factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by
the resolver and written by nothing.

A6/A7 — matchupKeys resolves those keys from the posted lineup plus the
schedule's probable pitchers, and fires the factors into a SHADOW freeze:
248 fires on 308 props, 245 of which would move the grade. The served
forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test
that decides whether they ever go live.

Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay
withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity
harness 34/34.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-11 22:49:56 -04:00

155 lines
7.5 KiB
JavaScript

#!/usr/bin/env node
'use strict';
/**
* proven-status — WHAT IS ACTUALLY PROVEN, computed from the ledger.
*
* WHY THIS EXISTS. Four consecutive build orders have opened by describing
* results as proven that the measurements did not support: "barrel rate PASSED
* solo" (every total_bases feature was refused on sample), "total_bases has
* passed BAR 1" (inconclusive at parity, CI spanning zero), "whiff/stuff prove
* SOLO through the gate" (refused at n=57), "two proven clusters live" (the
* proven set is empty). Each time the correction had to be re-derived by hand
* from a spec written days earlier.
*
* Prose decays. A number recomputed from the ledger does not. So this prints the
* proven set on demand, from the same gate everything else is held to, and any
* session can run it in one command before planning on top of a claim.
*
* IT DELIBERATELY CANNOT SAY "PROVEN" ON ITS OWN. A stat is proven only if a
* recorded head-to-head beat the counter out-of-sample with a CI excluding zero,
* which is a measurement this script does not perform — it reports SAMPLE
* READINESS (can the gate even be run?) and the recorded verdicts, so the two
* are never confused again.
*
* SUPABASE_URL=... node scripts/proven-status.js
*/
require('dotenv').config();
const { createClient } = require('@supabase/supabase-js');
const cv = require('../src/services/model/correlateValidator');
const { paginate } = require('../src/utils/safePaginate');
const { uniqueKeyFor } = require('../src/utils/tableKeys');
const SB_URL = process.env.SUPABASE_URL;
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
const PAGE = 1000;
const MIN_N = cv.VALIDATION_REQUIREMENTS.min_historical_instances;
/**
* RECORDED VERDICTS — every head-to-head this programme has actually run, with
* its spec. Add a row when a head-to-head is run; never edit one to be kinder.
*/
const RECORDED = [
{ stat: 'hits', n: 803, model: 0.0842, counter: 0.1803, delta: -0.0961, ci: [-0.1648, -0.0285],
verdict: 'LOSES', spec: 'specs/batter-cluster-prove.md' },
{ stat: 'total_bases', n: 383, model: 0.2685, counter: 0.2647, delta: 0.0038, ci: [-0.0675, 0.0753],
verdict: 'INCONCLUSIVE', spec: 'specs/tb-solo-and-interactions.md' },
{ stat: 'strikeouts', n: 57, model: 0.1953, counter: -0.0639, delta: 0.2592, ci: [-0.0167, 0.5645],
verdict: 'INCONCLUSIVE', spec: 'specs/lineup-k-rate-rung1.md' },
];
// ── FIX A2 (2026-08-09) — THE SAFE WALK ───────────────────────────────────
// `page()` above walks with no ORDER BY. Measured on production, that returned
// the correct row COUNT and the wrong ROWS: up to 33.6% of a read came back
// twice while an equal share never came back at all. `pageSafe` routes the same
// call through `src/utils/safePaginate`, which orders on a UNIQUE key on every
// page, verifies uniqueness at runtime, and THROWS on a query error instead of
// treating it as end-of-data.
//
// `page()` SURVIVES only for the context tables (statcast_aggregates,
// batter_spray, team_defense, platoon_splits, park_dimensions, game_context...).
// Those have COMPOSITE primary keys with no single unique column, so
// safePaginate cannot express them. They measure 0% corruption today; making
// them safe needs a composite-key ordering the helper does not yet have. Do not
// use `page()` for ledger_entries or model_snapshots.
async function pageSafe(sb, table, select, apply, key = uniqueKeyFor(table)) {
return paginate(() => apply(sb.from(table).select(select)),
{ key, pageSize: PAGE, label: `${table}` });
}
/** THE MEASURED READS — main() and the harness call the same functions. */
const READS_LEDGER = {
ledgerAll: (sb) => pageSafe(sb, 'ledger_entries', 'id, stat, outcome, quarantine_reason, p_win',
(q) => q.eq('sport', 'mlb').is('user_id', null)),
ledgerSettled: (sb) => pageSafe(sb, 'ledger_entries',
'id, stat, outcome, quarantine_reason, player_key, line, side, game_date',
(q) => q.eq('sport', 'mlb').is('user_id', null).in('outcome', ['hit', 'miss'])),
};
/** THE MEASURED READS — main() and the harness call the same functions. */
const READS = {
snaps: (sb) => pageSafe(sb, 'model_snapshots', 'id, stat, archetype, player_key, line, side, game_date',
(q) => q.eq('sport', 'mlb').not('archetype', 'is', null)),
};
async function main() {
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
const led = await pageSafe(sb, 'ledger_entries', 'id, stat, outcome, quarantine_reason, p_win',
(q) => q.eq('sport', 'mlb').is('user_id', null));
const settled = {};
for (const r of led) {
if ((r.quarantine_reason || '').startsWith('nontakeable_book')) continue;
if (r.outcome !== 'hit' && r.outcome !== 'miss') continue;
if (r.p_win == null) continue;
settled[r.stat] = (settled[r.stat] || 0) + 1;
}
const snaps = await READS.snaps(sb);
const archOf = new Map();
for (const s of snaps) archOf.set(`${s.player_key}|${s.stat}|${s.line}|${String(s.side).toLowerCase()}|${s.game_date}`, s.archetype);
// COUNT DISTINCT LEDGER ROWS. `model_snapshots` holds one row per prop PER
// SNAPSHOT CYCLE, so a naive join fans out and inflates the count — it read
// BOMBER x hits as 641 when the true figure is 287, which is the difference
// between "gate-ready" and "not close". Dedupe on the ledger row's identity.
const led2 = await pageSafe(sb, 'ledger_entries', 'id, stat, outcome, quarantine_reason, player_key, line, side, game_date',
(q) => q.eq('sport', 'mlb').is('user_id', null).in('outcome', ['hit', 'miss']));
const byArch = {};
const seen = new Set();
for (const r of led2) {
if ((r.quarantine_reason || '').startsWith('nontakeable_book')) continue;
if (seen.has(r.id)) continue;
seen.add(r.id);
const a = archOf.get(`${r.player_key}|${r.stat}|${r.line}|${String(r.side).toLowerCase()}|${r.game_date}`);
if (!a) continue;
const k = `${a} x ${r.stat}`;
byArch[k] = (byArch[k] || 0) + 1;
}
const gateReady = Object.entries(settled).filter(([, n]) => n >= MIN_N).map(([s, n]) => ({ stat: s, n }));
const archReady = Object.entries(byArch).filter(([, n]) => n >= MIN_N)
.sort((a, b) => b[1] - a[1]).map(([k, n]) => ({ combo: k, n }));
const proven = RECORDED.filter((r) => r.verdict === 'BEATS');
console.log(JSON.stringify({
generated_at_note: 'computed from the ledger; prose in specs may lag this',
gate_spec: cv.VALIDATION_REQUIREMENTS,
PROVEN_SET: proven.length === 0 ? 'EMPTY — no stat has beaten the counter out-of-sample with a CI excluding zero' : proven,
recorded_head_to_heads: RECORDED,
sample_readiness: {
note: 'n >= 500 means the gate CAN be run — it does not mean anything passed it',
stats_at_or_above_gate: gateReady,
stats_below_gate: Object.entries(settled).filter(([, n]) => n < MIN_N)
.sort((a, b) => b[1] - a[1]).map(([s, n]) => ({ stat: s, n, short_by: MIN_N - n })),
archetype_x_stat_at_or_above_gate: archReady,
archetype_x_stat_closest_below: Object.entries(byArch).filter(([, n]) => n < MIN_N)
.sort((a, b) => b[1] - a[1]).slice(0, 6).map(([k, n]) => ({ combo: k, n, short_by: MIN_N - n })),
},
}, null, 2));
process.exit(0);
}
if (require.main === module) {
main().catch((e) => { console.error(e); process.exit(1); });
}
// Exported so the read-integrity harness measures THE REAL FUNCTION.
module.exports = { READS, READS_LEDGER };