Files
vyndr/scripts/backfill-context.js
T
builtbykev f61ec6b391 Read integrity, as-of context, and the shadow matchup resolve (A1-A7)
Seven orders of measurement-first repair. The served grade does not move.

A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS:
410-617 of 2,490 duplicated with an equal number never returned, while
rows.length matched the server exactly. safePaginate orders on a real unique
key, verifies the tuple at runtime, and THROWS on a query error instead of
treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn
through that reader, and defense_by_direction's distinct-n was likely below
the gate floor all along.

A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from
pg_index (the context tables are dated-composite and had no single unique
column). The unordered helper is deleted, not parked.

A3 — ledgerService and retentionService defaulted the SAME env var to
DIFFERENT versions, so no ledger row ever carried the marker eligibility
requires. One source now. model_snapshots settlement moved onto the cron:
15,484 -> 28,894 settled, repaired-champion 0 -> 7,556.

A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no
row at-or-before the date means the factor does not apply, never the nearest
row. Live path unchanged, proven 400/400 on real rows.

A5 — factor_inputs freezes what the factor READ, never the multiplier, so an
audit can recompute and check. It also recorded the finding: the three hits
factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by
the resolver and written by nothing.

A6/A7 — matchupKeys resolves those keys from the posted lineup plus the
schedule's probable pitchers, and fires the factors into a SHADOW freeze:
248 fires on 308 props, 245 of which would move the grade. The served
forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test
that decides whether they ever go live.

Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay
withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity
harness 34/34.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-11 22:49:56 -04:00

124 lines
6.0 KiB
JavaScript

#!/usr/bin/env node
'use strict';
/**
* backfill-context — WERE WE WAITING, OR UNDER-QUERYING?
*
* The platoon test ran on 452 rows against 1,266 clean settled hits rows in the
* ledger, so "48 short of the gate" was never a statement about how much data
* exists. It was a statement about how much the JOIN survived — and the join was
* losing rows to inputs we simply had not fetched for every player.
*
* This backfills the inputs (pure sample, zero waiting) and reports exactly
* where each row is lost, so the next "we need more data" claim is a measured
* one rather than an inherited one.
*
* ── THE ONE HONEST CAVEAT, STATED UP FRONT ───────────────────────────────
* Platoon splits from statsapi are SEASON-TO-DATE as of the moment they are
* fetched. Applying today's split to a 2026-07-15 game means the split contains
* that game. For a ~400-PA season line one game is roughly a quarter of one
* percent, so the contamination is small — but it is real, it runs in the
* flattering direction, and it is why this is labelled a reconstruction rather
* than a clean point-in-time backtest.
*
* SUPABASE_URL=... node scripts/backfill-context.js
*/
require('dotenv').config();
const { createClient } = require('@supabase/supabase-js');
const ctx = require('../src/services/lineupContextService');
const mlb = require('../src/services/adapters/mlbStatsAdapter');
const { knownNumber } = require('../src/utils/known');
const { paginate } = require('../src/utils/safePaginate');
const { uniqueKeyFor } = require('../src/utils/tableKeys');
const SB_URL = process.env.SUPABASE_URL;
const SB_KEY = process.env.SUPABASE_SERVICE_ROLE_KEY || process.env.SUPABASE_SERVICE_KEY;
const SEASON = Number(process.env.BF_SEASON || 2026);
const PAGE = 1000;
// ── THE SAFE WALK (Fix A2/A2b) ────────────────────────────────────────────
// This script used to walk pages with `.range()` and NO ORDER BY. Measured on
// production, that returned the correct row COUNT and the wrong ROWS: up to
// 33.6% of a read came back twice while an equal share never came back at all,
// so `rows.length` looked perfect while a fifth of the sample was missing.
//
// `pageSafe` routes every read through `src/utils/safePaginate`: a stable ORDER
// BY on the table's real UNIQUE key — single OR composite, looked up from
// `src/utils/tableKeys` rather than assumed — a runtime tuple-uniqueness check,
// and a THROWN error instead of a silent stop. The old unordered helper is gone
// rather than left beside it, because a dead broken helper is an invitation.
async function pageSafe(sb, table, select, apply, key = uniqueKeyFor(table)) {
return paginate(() => apply(sb.from(table).select(select)),
{ key, pageSize: PAGE, label: `${table}` });
}
/** THE MEASURED READS — main() and the harness call the same functions. */
const READS = {
ledger: (sb) => pageSafe(sb, 'ledger_entries', 'id, player_key, player_name, stat, outcome, quarantine_reason',
(q) => q.eq('sport', 'mlb').is('user_id', null).in('stat', ['hits', 'total_bases'])
.in('outcome', ['hit', 'miss'])),
platoonHeld: (sb) => pageSafe(sb, 'platoon_splits', 'as_of_date, sport, season, player_key',
(q) => q.eq('sport', 'mlb')),
};
async function main() {
if (!SB_URL || !SB_KEY) throw new Error('SUPABASE_URL / service key required');
const sb = createClient(SB_URL, SB_KEY, { auth: { persistSession: false } });
// Every hitter who appears on a CLEAN settled row — the true denominator.
const led = await pageSafe(sb, 'ledger_entries', 'id, player_key, player_name, stat, outcome, quarantine_reason',
(q) => q.eq('sport', 'mlb').is('user_id', null).in('stat', ['hits', 'total_bases'])
.in('outcome', ['hit', 'miss']));
const need = new Map();
for (const r of led) {
if ((r.quarantine_reason || '').startsWith('nontakeable_book')) continue;
if (!need.has(r.player_key)) need.set(r.player_key, r.player_name);
}
const have = new Set((await pageSafe(sb, 'platoon_splits', 'as_of_date, sport, season, player_key', (q) => q.eq('sport', 'mlb')))
.map((r) => r.player_key));
const missing = [...need.entries()].filter(([k]) => !have.has(k));
console.error(`[backfill] hitters on clean settled rows: ${need.size}; splits already held: ${have.size}; to fetch: ${missing.length}`);
const asOf = ctx.dateET();
const rows = [];
let unresolved = 0;
for (const [key, name] of missing) {
let found = null;
try { found = await mlb.searchPlayer(name); } catch { found = null; }
if (!found || !found.id) { unresolved += 1; continue; }
const sp = await ctx.fetchPlatoonSplits(found.id, SEASON, {});
if (!sp) continue; // absent, never a symmetric guess
rows.push({ player_key: key, player_name: name, source_id: found.id, ...sp });
}
let written = 0;
for (let i = 0; i < rows.length; i += 200) {
const batch = rows.slice(i, i + 200).map((r) => ({ ...r, sport: 'mlb', season: SEASON, as_of_date: asOf }));
const { error } = await sb.from('platoon_splits')
.upsert(batch, { onConflict: 'as_of_date,sport,season,player_key' });
if (!error) written += batch.length;
else console.error('[backfill] write failed:', error.message);
}
console.log(JSON.stringify({
hitters_on_clean_settled_rows: need.size,
splits_held_before: have.size,
attempted: missing.length,
unresolved_by_name: unresolved,
no_splits_available: missing.length - unresolved - rows.length,
written,
caveat: 'season-to-date splits applied to past games contain those games — small (~0.25% of a 400-PA line) but real and flattering',
}, null, 2));
process.exit(0);
}
if (require.main === module) {
main().catch((e) => { console.error(e); process.exit(1); });
}
// Exported so the read-integrity harness measures THE REAL FUNCTION.
module.exports = { READS: (typeof READS !== 'undefined' ? READS : undefined), READS_LEDGER: (typeof READS_LEDGER !== 'undefined' ? READS_LEDGER : undefined) };