f61ec6b391
Seven orders of measurement-first repair. The served grade does not move. A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS: 410-617 of 2,490 duplicated with an equal number never returned, while rows.length matched the server exactly. safePaginate orders on a real unique key, verifies the tuple at runtime, and THROWS on a query error instead of treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn through that reader, and defense_by_direction's distinct-n was likely below the gate floor all along. A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from pg_index (the context tables are dated-composite and had no single unique column). The unordered helper is deleted, not parked. A3 — ledgerService and retentionService defaulted the SAME env var to DIFFERENT versions, so no ledger row ever carried the marker eligibility requires. One source now. model_snapshots settlement moved onto the cron: 15,484 -> 28,894 settled, repaired-champion 0 -> 7,556. A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no row at-or-before the date means the factor does not apply, never the nearest row. Live path unchanged, proven 400/400 on real rows. A5 — factor_inputs freezes what the factor READ, never the multiplier, so an audit can recompute and check. It also recorded the finding: the three hits factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by the resolver and written by nothing. A6/A7 — matchupKeys resolves those keys from the posted lineup plus the schedule's probable pitchers, and fires the factors into a SHADOW freeze: 248 fires on 308 props, 245 of which would move the grade. The served forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test that decides whether they ever go live. Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity harness 34/34. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
149 lines
6.7 KiB
JavaScript
149 lines
6.7 KiB
JavaScript
#!/usr/bin/env node
|
|
'use strict';
|
|
|
|
/**
|
|
* calibrate-hits — FIT PAST, APPLY FORWARD, VERIFY HELD-OUT.
|
|
*
|
|
* The parlay surface is blocked because hit probabilities are well-ranked and
|
|
* badly calibrated: the model claims 0.911 and realises 0.630, and it is flat
|
|
* above 0.70. Compounding multiplies that error, so the repair has to be proven
|
|
* on data the correction never saw.
|
|
*
|
|
* THE ONE DISCIPLINE THAT MAKES THIS MEAN ANYTHING: the map is fitted on an
|
|
* EARLIER window and evaluated on a LATER one. Fitting and evaluating on the
|
|
* same rows always looks perfectly calibrated — that is not a result, it is the
|
|
* map reciting the answers it was built from. Any calibration report that does
|
|
* not name its split should be assumed to have done exactly that.
|
|
*
|
|
* WHY ISOTONIC. It is monotone by construction, so the model's ORDERING survives
|
|
* untouched and only the magnitudes move. We are repairing what it counts, not
|
|
* what it ranks — and the ranking is the part that measured well.
|
|
*
|
|
* PASS CONDITION: the TOP BINS (0.70+) must be honest out-of-sample. A parlay is
|
|
* built from confident legs, so calibration that only holds in the middle is
|
|
* worthless for the thing this unblocks.
|
|
*
|
|
* SUPABASE_URL=... node scripts/calibrate-hits.js
|
|
*/
|
|
|
|
require('dotenv').config();
|
|
const { createClient } = require('@supabase/supabase-js');
|
|
const cal = require('../src/services/model/calibration');
|
|
const { paginate } = require('../src/utils/safePaginate');
|
|
const { uniqueKeyFor } = require('../src/utils/tableKeys');
|
|
|
|
const SB_URL = process.env.SUPABASE_URL;
|
|
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
|
|
const SPLIT = process.env.CAL_SPLIT || '2026-08-02'; // held-out starts here
|
|
const PAGE = 1000;
|
|
|
|
const r3 = (v) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 1000) / 1000);
|
|
|
|
// ── THE SAFE WALK (Fix A2b) ───────────────────────────────────────────────
|
|
// This walked pages with no ORDER BY and measured 24.7% corrupt on production —
|
|
// 616 of 2,490 rows returned twice, an equal share never returned — while
|
|
// `rows.length` matched the server count exactly. It fits the calibration
|
|
// reliability curve, so a fifth of the history was double-weighted and another
|
|
// fifth absent from every bin.
|
|
async function pageSafe(sb, apply) {
|
|
return paginate(() => apply(sb.from('ledger_entries')
|
|
.select('id, p_win, outcome, game_date, quarantine_reason')),
|
|
{ key: uniqueKeyFor('ledger_entries'), pageSize: PAGE, label: 'calibrate-hits' });
|
|
}
|
|
|
|
/** THE MEASURED READ — main() and the harness call the same function. */
|
|
const READS = {
|
|
ledger: (sb) => pageSafe(sb, (q) => q.eq('sport', 'mlb').is('user_id', null).eq('stat', 'hits')
|
|
.in('outcome', ['hit', 'miss']).not('p_win', 'is', null)),
|
|
};
|
|
|
|
/** Reliability rendered per bin with n — the only honest way to read this. */
|
|
function curve(rows, label) {
|
|
return cal.reliability(rows, 10)
|
|
.filter((b) => b.n >= 10)
|
|
.map((b) => ({
|
|
window: label,
|
|
predicted: r3(b.mean_predicted),
|
|
actual: r3(b.actual),
|
|
error: r3(b.error),
|
|
n: b.n,
|
|
}));
|
|
}
|
|
|
|
async function main() {
|
|
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
|
|
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
|
|
|
|
const raw = await READS.ledger(sb);
|
|
const all = raw
|
|
.filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book'))
|
|
.map((r) => ({ p: Number(r.p_win), won: r.outcome === 'hit' ? 1 : 0, d: String(r.game_date) }));
|
|
|
|
const fit = all.filter((r) => r.d < SPLIT);
|
|
const held = all.filter((r) => r.d >= SPLIT);
|
|
|
|
const map = cal.fitIsotonic(fit);
|
|
if (!map) {
|
|
console.log(JSON.stringify({ ok: false, reason: 'could not fit', fit_n: fit.length }));
|
|
process.exit(0);
|
|
}
|
|
|
|
// Apply the FIT-WINDOW map to the HELD-OUT rows. The map has never seen these.
|
|
const corrected = held.map((r) => ({ ...r, p: cal.applyIsotonic(map, r.p) }));
|
|
|
|
const before = cal.isCalibrated(held, { minTotal: 100 });
|
|
const after = cal.isCalibrated(corrected, { minTotal: 100 });
|
|
|
|
// Did the ORDERING survive? Isotonic is monotone, so it must — checked rather
|
|
// than asserted, because a broken map would silently destroy the one thing
|
|
// the model does well.
|
|
const pairs = [];
|
|
for (let i = 0; i < Math.min(held.length, 400); i += 1) {
|
|
for (let j = i + 1; j < Math.min(held.length, 400); j += 1) {
|
|
if (held[i].p === held[j].p) continue;
|
|
const rawOrder = Math.sign(held[i].p - held[j].p);
|
|
const calOrder = Math.sign(corrected[i].p - corrected[j].p);
|
|
pairs.push(calOrder === 0 || calOrder === rawOrder);
|
|
}
|
|
}
|
|
const orderingPreserved = pairs.length === 0 || pairs.every(Boolean);
|
|
|
|
const topBefore = curve(held, 'held-out RAW').filter((b) => b.predicted >= 0.70);
|
|
// NOT ">= 0.70": honest calibration REMOVES the 0.70+ predictions entirely
|
|
// (the ceiling drops to ~0.667), so demanding that band exist would fail the
|
|
// map for succeeding. The right question is whether the model's HIGHEST
|
|
// REMAINING confidence band is honest, because that is what a parlay stacks.
|
|
const afterCurve = curve(corrected, 'held-out CALIBRATED');
|
|
const topAfter = afterCurve.slice(-2);
|
|
const bands = cal.certifyBands(corrected, { tolerance: 0.05, minBin: 40 });
|
|
const ceiling = afterCurve.length ? Math.max(...afterCurve.map((b) => b.predicted)) : null;
|
|
|
|
console.log(JSON.stringify({
|
|
discipline: `fitted on game_date < ${SPLIT}, evaluated on game_date >= ${SPLIT} — the map never saw the evaluation rows`,
|
|
fit_n: fit.length,
|
|
held_out_n: held.length,
|
|
ordering_preserved: orderingPreserved,
|
|
isotonic_blocks: map.length,
|
|
map_sample: [0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 0.95].map((p) => ({ claims: p, corrected_to: r3(cal.applyIsotonic(map, p)) })),
|
|
held_out_before: { calibrated: before.calibrated, max_bin_error: r3(before.max_bin_error), curve: curve(held, 'RAW') },
|
|
held_out_after: { calibrated: after.calibrated, max_bin_error: r3(after.max_bin_error), curve: curve(corrected, 'CALIBRATED') },
|
|
top_bins_before: topBefore,
|
|
top_bins_after: topAfter,
|
|
certified_bands: bands,
|
|
honest_ceiling: ceiling,
|
|
four_leg_ticket_at_ceiling: ceiling ? r3(ceiling ** 4) : null,
|
|
verdict: !orderingPreserved ? 'FAIL — ordering destroyed'
|
|
: bands.length === 0 ? 'FAIL — no band is honest out-of-sample'
|
|
: `PARTIAL PASS — honest within ${bands.map((b) => `${b.lo}-${b.hi}`).join(', ')}; outside those bands legs are NOT stackable`,
|
|
}, null, 2));
|
|
process.exit(0);
|
|
}
|
|
|
|
if (require.main === module) {
|
|
main().catch((e) => { console.error(e); process.exit(1); });
|
|
}
|
|
|
|
// Exported so the read-integrity harness measures THE REAL FUNCTION.
|
|
module.exports = { READS };
|
|
|