Scoreboard every challenger; diagnose the 429 as odds-api, not PropLine
PROMOTE-THE-EARNED. Nothing was promoted, because nothing earned it -- not because the bar was held high. Measured on the same bar that refuted hits-v1: own rows only, direction-aligned, paired bootstrap, promote only on a CI excluding zero. arch-v1 n=1741 delta 0.0000 CI[-0.0050,+0.0054] inconclusive contact-v1 n=1055 delta +0.0008 CI[-0.0052,+0.0069] inconclusive proj-v1.1 n=1664 delta -0.0301 CI[-0.0543,-0.0060] reliably WORSE matchup/tb-v1/hits-v1 n=0 genuinely pending (rows dated 08-02+) arch-v1 is the interesting one: it MOVED 76% of rows by 2.5 points on average and resolution is identical to the champion to four decimals, on the moved rows too. That is active movement carrying no information -- a finding, not a pending verdict. These are true prospective holdouts: arch-v1 and contact-v1 wrote p_win at grade time into their own columns before the game. Nothing recomputed. THE 429, read-only. The premise was that we re-pull the full picture every slot and blow the quota. Measured: PropLine is at 5 calls of 3,000/day -- 0.17%. One snapshot is ONE PropLine call per sport, all markets comma-joined. There is no request-pattern problem, so a change-based pull cannot fix it and no tier upgrade is needed. The 429 is odds-api: 478/500 MONTHLY, blocked at 95%. oddsService falls through silently when PropLine returns empty, and the backup's quota gate throws the error -- so an empty slate is indistinguishable from an outage and the message names the wrong provider. Flagged for its own order. Could NOT verify PropLine movement endpoints: docs are auth-gated and the keys are production-only. Not asserted either way. The movement-as-data argument stands on its own merits and should be justified that way, not as a quota fix it isn't. Book-breadth invariant written down: we never discard books. All are kept and shown (DISPLAY_BOOKS = MODEL + REFERENCE + DFS); DFS pick'em is excluded from PRICING only, because a fixed-payout shaded number is not a market price. Verified this is already what bookRoles.js does. Champion byte-identical; every challenger stays wired. 4,159 tests green (332 suites); web build exit 0. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01W1sivYNqY2TS5ftykmHBU9
This commit is contained in:
@@ -0,0 +1,191 @@
|
||||
#!/usr/bin/env node
|
||||
'use strict';
|
||||
|
||||
/**
|
||||
* challenger-scoreboard — every accruing challenger, measured on the same bar.
|
||||
*
|
||||
* WHY THIS IS A REAL HOLDOUT AND NOT A BACKTEST. Unlike hits-v1 (which did not
|
||||
* exist when these rows were graded and therefore needed a point-in-time
|
||||
* replay), arch-v1 and contact-v1 wrote their probability AT GRADE TIME, into
|
||||
* their own columns, before the game was played. Nothing here is recomputed.
|
||||
* These numbers are genuinely out-of-sample — the strongest evidence available.
|
||||
*
|
||||
* THE BAR IS THE SAME ONE THAT REFUTED hits-v1:
|
||||
* - the challenger's OWN rows only (a challenger that abstains is not scored
|
||||
* on the rows it declined — averaging those in measures the champion twice)
|
||||
* - direction handled: p_win and p_win_challenger/p_win_contact are all
|
||||
* P(GRADED SIDE), so they are already aligned. The projection ladder is
|
||||
* P(OVER) and IS realigned here.
|
||||
* - paired bootstrap on matched rows, because both models score the SAME rows
|
||||
* and treating their errors as independent overstates the uncertainty
|
||||
* - PROMOTE only when the CI on (challenger − champion) excludes zero
|
||||
*
|
||||
* CONTAMINATION EXCLUSION: rows whose price/book were stamped from a
|
||||
* non-takeable book are tagged `quarantine_reason LIKE 'nontakeable_book%'` and
|
||||
* are excluded — their locked price describes a market you could not have bet.
|
||||
*
|
||||
* PROVENANCE: results are reported for all rows AND split by `model_version`,
|
||||
* so if a verdict depends on the older `pre-retention-unknown` era that fact is
|
||||
* visible rather than buried.
|
||||
*
|
||||
* SUPABASE_URL=... node scripts/challenger-scoreboard.js
|
||||
*/
|
||||
|
||||
require('dotenv').config();
|
||||
const { createClient } = require('@supabase/supabase-js');
|
||||
|
||||
const SB_URL = process.env.SUPABASE_URL;
|
||||
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
|
||||
const PAGE = 1000;
|
||||
|
||||
function corr(xs, ys) {
|
||||
const n = xs.length;
|
||||
if (n < 3) return null;
|
||||
const mx = xs.reduce((a, b) => a + b, 0) / n;
|
||||
const my = ys.reduce((a, b) => a + b, 0) / n;
|
||||
let sxy = 0; let sxx = 0; let syy = 0;
|
||||
for (let i = 0; i < n; i += 1) {
|
||||
const dx = xs[i] - mx; const dy = ys[i] - my;
|
||||
sxy += dx * dy; sxx += dx * dx; syy += dy * dy;
|
||||
}
|
||||
if (sxx <= 0 || syy <= 0) return null;
|
||||
return sxy / Math.sqrt(sxx * syy);
|
||||
}
|
||||
const r4 = (v) => (v == null ? null : Math.round(v * 10000) / 10000);
|
||||
const meanOf = (a) => (a.length ? a.reduce((x, y) => x + y, 0) / a.length : null);
|
||||
const brier = (ps, ys) => (ps.length ? ps.reduce((s, p, i) => s + (p - ys[i]) ** 2, 0) / ps.length : null);
|
||||
|
||||
/** Paired bootstrap on the DIFFERENCE of resolutions. Deterministic seed. */
|
||||
function bootstrapDiff(rows, keyA, keyB, iters = 4000, seed = 20260803) {
|
||||
if (rows.length < 30) return null;
|
||||
let s = seed >>> 0;
|
||||
const rnd = () => { s ^= s << 13; s >>>= 0; s ^= s >>> 17; s ^= s << 5; s >>>= 0; return s / 4294967296; };
|
||||
const n = rows.length;
|
||||
const diffs = [];
|
||||
for (let it = 0; it < iters; it += 1) {
|
||||
const ys = []; const a = []; const b = [];
|
||||
for (let i = 0; i < n; i += 1) {
|
||||
const r = rows[Math.floor(rnd() * n)];
|
||||
ys.push(r.won); a.push(r[keyA]); b.push(r[keyB]);
|
||||
}
|
||||
const ca = corr(a, ys); const cb = corr(b, ys);
|
||||
if (ca == null || cb == null) continue;
|
||||
diffs.push(ca - cb);
|
||||
}
|
||||
if (diffs.length < 100) return null;
|
||||
diffs.sort((x, y) => x - y);
|
||||
const q = (p) => r4(diffs[Math.floor(p * (diffs.length - 1))]);
|
||||
const point = r4(corr(rows.map((r) => r[keyA]), rows.map((r) => r.won))
|
||||
- corr(rows.map((r) => r[keyB]), rows.map((r) => r.won)));
|
||||
const ci = [q(0.025), q(0.975)];
|
||||
return { point, ci95: ci, p_improves: r4(diffs.filter((d) => d > 0).length / diffs.length),
|
||||
ci_excludes_zero: ci[0] > 0 || ci[1] < 0 };
|
||||
}
|
||||
|
||||
function score(rows, challKey, label) {
|
||||
const ys = rows.map((r) => r.won);
|
||||
const ch = rows.map((r) => r[challKey]);
|
||||
const cp = rows.map((r) => r.champ);
|
||||
const bs = bootstrapDiff(rows, challKey, 'champ');
|
||||
let verdict = 'STILL PENDING';
|
||||
if (rows.length >= 30 && bs) {
|
||||
if (bs.ci_excludes_zero && bs.point > 0) verdict = 'PROMOTE';
|
||||
else if (bs.ci_excludes_zero && bs.point < 0) verdict = 'STAY WIRED (measured worse)';
|
||||
else verdict = 'STAY WIRED (inconclusive)';
|
||||
}
|
||||
return {
|
||||
challenger: label,
|
||||
settled_n: rows.length,
|
||||
base_rate: r4(meanOf(ys)),
|
||||
resolution_challenger: r4(corr(ch, ys)),
|
||||
resolution_champion: r4(corr(cp, ys)),
|
||||
brier_challenger: r4(brier(ch, ys)),
|
||||
brier_champion: r4(brier(cp, ys)),
|
||||
delta_vs_champion: bs,
|
||||
verdict,
|
||||
};
|
||||
}
|
||||
|
||||
async function fetchAll(sb) {
|
||||
const out = [];
|
||||
for (let from = 0; ; from += PAGE) {
|
||||
const { data, error } = await sb.from('ledger_entries')
|
||||
.select('id, stat, side, outcome, model_version, quarantine_reason, p_win, p_win_challenger, p_win_contact, challenger_delta, contact_delta, proj_p_over_line, proj_tb_p_over, proj_hits_p_over, challenger_adjustments')
|
||||
.eq('sport', 'mlb').is('user_id', null)
|
||||
.in('outcome', ['hit', 'miss'])
|
||||
.not('p_win', 'is', null)
|
||||
.range(from, from + PAGE - 1);
|
||||
if (error) throw error;
|
||||
if (!data || data.length === 0) break;
|
||||
out.push(...data);
|
||||
if (data.length < PAGE) break;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
async function main() {
|
||||
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
|
||||
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
|
||||
|
||||
const raw = (await fetchAll(sb))
|
||||
.filter((r) => !(r.quarantine_reason || '').startsWith('nontakeable_book'));
|
||||
|
||||
const base = raw.map((r) => ({
|
||||
won: r.outcome === 'hit' ? 1 : 0,
|
||||
champ: Number(r.p_win),
|
||||
arch: r.p_win_challenger == null ? null : Number(r.p_win_challenger),
|
||||
contact: r.p_win_contact == null ? null : Number(r.p_win_contact),
|
||||
// The ladder is P(OVER); realign it to the graded side before comparing.
|
||||
ladder: r.proj_p_over_line == null ? null
|
||||
: (String(r.side).toLowerCase() === 'under' ? 1 - Number(r.proj_p_over_line) : Number(r.proj_p_over_line)),
|
||||
tb: r.proj_tb_p_over == null ? null
|
||||
: (String(r.side).toLowerCase() === 'under' ? 1 - Number(r.proj_tb_p_over) : Number(r.proj_tb_p_over)),
|
||||
hits: r.proj_hits_p_over == null ? null
|
||||
: (String(r.side).toLowerCase() === 'under' ? 1 - Number(r.proj_hits_p_over) : Number(r.proj_hits_p_over)),
|
||||
stat: r.stat,
|
||||
// Did the challenger actually MOVE this row? A nudge that leaves p_win
|
||||
// untouched is the champion wearing a different name, and scoring it on
|
||||
// those rows measures the champion against itself — which is exactly how a
|
||||
// real effect gets averaged down to zero.
|
||||
archMoved: r.challenger_delta != null && Number(r.challenger_delta) !== 0,
|
||||
contactMoved: r.contact_delta != null && Number(r.contact_delta) !== 0,
|
||||
era: r.model_version || 'unknown',
|
||||
axes: new Set(((r.challenger_adjustments) || []).map((a) => a && a.axis).filter(Boolean)),
|
||||
}));
|
||||
|
||||
const withKey = (k, extra = () => true) => base.filter((r) => r[k] != null && Number.isFinite(r[k]) && extra(r));
|
||||
|
||||
const board = [
|
||||
score(withKey('arch'), 'arch', 'arch-v1 (market-relative nudge)'),
|
||||
score(withKey('contact'), 'contact', 'contact-v1 (season contact quality)'),
|
||||
score(withKey('ladder'), 'ladder', 'proj-v1.1 ladder (all stats)'),
|
||||
score(withKey('tb', (r) => r.stat === 'total_bases'), 'tb', 'tb-v1 (total_bases only)'),
|
||||
score(withKey('hits', (r) => r.stat === 'hits'), 'hits', 'hits-v1 (hits only)'),
|
||||
];
|
||||
|
||||
// THE SHARPEST TEST OF A NUDGE — only the rows it actually moved.
|
||||
const movedBoard = [
|
||||
score(withKey('arch', (r) => r.archMoved), 'arch', 'arch-v1 · rows it MOVED only'),
|
||||
score(withKey('contact', (r) => r.contactMoved), 'contact', 'contact-v1 · rows it MOVED only'),
|
||||
];
|
||||
|
||||
// Per-AXIS: arch-v1 restricted to the rows where that axis actually fired.
|
||||
const axisBoard = ['environment', 'opportunity', 'matchup'].map((ax) =>
|
||||
score(withKey('arch', (r) => r.axes.has(ax)), 'arch', `arch-v1 · ${ax} axis rows only`));
|
||||
|
||||
// PROVENANCE split — does any verdict depend on the older era?
|
||||
const eras = [...new Set(base.map((r) => r.era))];
|
||||
const provenance = eras.map((era) => ({
|
||||
era,
|
||||
...score(withKey('arch', (r) => r.era === era), 'arch', `arch-v1 · ${era}`),
|
||||
}));
|
||||
|
||||
console.log(JSON.stringify({
|
||||
measurement: 'PROSPECTIVE HOLDOUT — challenger values were written at grade time, before the game. No recomputation, no lookahead.',
|
||||
total_settled_rows: base.length,
|
||||
board, movedBoard, axisBoard, provenance,
|
||||
}, null, 2));
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
main().catch((e) => { console.error(e); process.exit(1); });
|
||||
Reference in New Issue
Block a user