Files
vyndr/scripts/calibrate-hits.js
T
builtbykev f61ec6b391 Read integrity, as-of context, and the shadow matchup resolve (A1-A7)
Seven orders of measurement-first repair. The served grade does not move.

A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS:
410-617 of 2,490 duplicated with an equal number never returned, while
rows.length matched the server exactly. safePaginate orders on a real unique
key, verifies the tuple at runtime, and THROWS on a query error instead of
treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn
through that reader, and defense_by_direction's distinct-n was likely below
the gate floor all along.

A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from
pg_index (the context tables are dated-composite and had no single unique
column). The unordered helper is deleted, not parked.

A3 — ledgerService and retentionService defaulted the SAME env var to
DIFFERENT versions, so no ledger row ever carried the marker eligibility
requires. One source now. model_snapshots settlement moved onto the cron:
15,484 -> 28,894 settled, repaired-champion 0 -> 7,556.

A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no
row at-or-before the date means the factor does not apply, never the nearest
row. Live path unchanged, proven 400/400 on real rows.

A5 — factor_inputs freezes what the factor READ, never the multiplier, so an
audit can recompute and check. It also recorded the finding: the three hits
factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by
the resolver and written by nothing.

A6/A7 — matchupKeys resolves those keys from the posted lineup plus the
schedule's probable pitchers, and fires the factors into a SHADOW freeze:
248 fires on 308 props, 245 of which would move the grade. The served
forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test
that decides whether they ever go live.

Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay
withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity
harness 34/34.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-11 22:49:56 -04:00

149 lines
6.7 KiB
JavaScript

#!/usr/bin/env node
'use strict';
/**
* calibrate-hits — FIT PAST, APPLY FORWARD, VERIFY HELD-OUT.
*
* The parlay surface is blocked because hit probabilities are well-ranked and
* badly calibrated: the model claims 0.911 and realises 0.630, and it is flat
* above 0.70. Compounding multiplies that error, so the repair has to be proven
* on data the correction never saw.
*
* THE ONE DISCIPLINE THAT MAKES THIS MEAN ANYTHING: the map is fitted on an
* EARLIER window and evaluated on a LATER one. Fitting and evaluating on the
* same rows always looks perfectly calibrated — that is not a result, it is the
* map reciting the answers it was built from. Any calibration report that does
* not name its split should be assumed to have done exactly that.
*
* WHY ISOTONIC. It is monotone by construction, so the model's ORDERING survives
* untouched and only the magnitudes move. We are repairing what it counts, not
* what it ranks — and the ranking is the part that measured well.
*
* PASS CONDITION: the TOP BINS (0.70+) must be honest out-of-sample. A parlay is
* built from confident legs, so calibration that only holds in the middle is
* worthless for the thing this unblocks.
*
* SUPABASE_URL=... node scripts/calibrate-hits.js
*/
require('dotenv').config();
const { createClient } = require('@supabase/supabase-js');
const cal = require('../src/services/model/calibration');
const { paginate } = require('../src/utils/safePaginate');
const { uniqueKeyFor } = require('../src/utils/tableKeys');
const SB_URL = process.env.SUPABASE_URL;
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
const SPLIT = process.env.CAL_SPLIT || '2026-08-02'; // held-out starts here
const PAGE = 1000;
const r3 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 1000) / 1000);
// ── THE SAFE WALK (Fix A2b) ───────────────────────────────────────────────
// This walked pages with no ORDER BY and measured 24.7% corrupt on production —
// 616 of 2,490 rows returned twice, an equal share never returned — while
// `rows.length` matched the server count exactly. It fits the calibration
// reliability curve, so a fifth of the history was double-weighted and another
// fifth absent from every bin.
async function pageSafe(sb, apply) {
return paginate(() => apply(sb.from('ledger_entries')
.select('id, p_win, outcome, game_date, quarantine_reason')),
{ key: uniqueKeyFor('ledger_entries'), pageSize: PAGE, label: 'calibrate-hits' });
}
/** THE MEASURED READ — main() and the harness call the same function. */
const READS = {
ledger: (sb) => pageSafe(sb, (q) => q.eq('sport', 'mlb').is('user_id', null).eq('stat', 'hits')
.in('outcome', ['hit', 'miss']).not('p_win', 'is', null)),
};
/** Reliability rendered per bin with n — the only honest way to read this. */
function curve(rows, label) {
return cal.reliability(rows, 10)
.filter((b) => b.n >= 10)
.map((b) => ({
window: label,
predicted: r3(b.mean_predicted),
actual: r3(b.actual),
error: r3(b.error),
n: b.n,
}));
}
async function main() {
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
const raw = await READS.ledger(sb);
const all = raw
.filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book'))
.map((r) => ({ p: Number(r.p_win), won: r.outcome === 'hit' ? 1 : 0, d: String(r.game_date) }));
const fit = all.filter((r) => r.d < SPLIT);
const held = all.filter((r) => r.d >= SPLIT);
const map = cal.fitIsotonic(fit);
if (!map) {
console.log(JSON.stringify({ ok: false, reason: 'could not fit', fit_n: fit.length }));
process.exit(0);
}
// Apply the FIT-WINDOW map to the HELD-OUT rows. The map has never seen these.
const corrected = held.map((r) => ({ ...r, p: cal.applyIsotonic(map, r.p) }));
const before = cal.isCalibrated(held, { minTotal: 100 });
const after = cal.isCalibrated(corrected, { minTotal: 100 });
// Did the ORDERING survive? Isotonic is monotone, so it must — checked rather
// than asserted, because a broken map would silently destroy the one thing
// the model does well.
const pairs = [];
for (let i = 0; i < Math.min(held.length, 400); i += 1) {
for (let j = i + 1; j < Math.min(held.length, 400); j += 1) {
if (held[i].p === held[j].p) continue;
const rawOrder = Math.sign(held[i].p - held[j].p);
const calOrder = Math.sign(corrected[i].p - corrected[j].p);
pairs.push(calOrder === 0 || calOrder === rawOrder);
}
}
const orderingPreserved = pairs.length === 0 || pairs.every(Boolean);
const topBefore = curve(held, 'held-out RAW').filter((b) => b.predicted >= 0.70);
// NOT ">= 0.70": honest calibration REMOVES the 0.70+ predictions entirely
// (the ceiling drops to ~0.667), so demanding that band exist would fail the
// map for succeeding. The right question is whether the model's HIGHEST
// REMAINING confidence band is honest, because that is what a parlay stacks.
const afterCurve = curve(corrected, 'held-out CALIBRATED');
const topAfter = afterCurve.slice(-2);
const bands = cal.certifyBands(corrected, { tolerance: 0.05, minBin: 40 });
const ceiling = afterCurve.length ? Math.max(...afterCurve.map((b) => b.predicted)) : null;
console.log(JSON.stringify({
discipline: `fitted on game_date < ${SPLIT}, evaluated on game_date >= ${SPLIT} — the map never saw the evaluation rows`,
fit_n: fit.length,
held_out_n: held.length,
ordering_preserved: orderingPreserved,
isotonic_blocks: map.length,
map_sample: [0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 0.95].map((p) => ({ claims: p, corrected_to: r3(cal.applyIsotonic(map, p)) })),
held_out_before: { calibrated: before.calibrated, max_bin_error: r3(before.max_bin_error), curve: curve(held, 'RAW') },
held_out_after: { calibrated: after.calibrated, max_bin_error: r3(after.max_bin_error), curve: curve(corrected, 'CALIBRATED') },
top_bins_before: topBefore,
top_bins_after: topAfter,
certified_bands: bands,
honest_ceiling: ceiling,
four_leg_ticket_at_ceiling: ceiling ? r3(ceiling ** 4) : null,
verdict: !orderingPreserved ? 'FAIL — ordering destroyed'
: bands.length === 0 ? 'FAIL — no band is honest out-of-sample'
: `PARTIAL PASS — honest within ${bands.map((b) => `${b.lo}-${b.hi}`).join(', ')}; outside those bands legs are NOT stackable`,
}, null, 2));
process.exit(0);
}
if (require.main === module) {
main().catch((e) => { console.error(e); process.exit(1); });
}
// Exported so the read-integrity harness measures THE REAL FUNCTION.
module.exports = { READS };