checkpoint: chain shadow, WNBA possession feed, baseball chain

Backup commit of uncommitted working-tree state found during Legion
recon (Tony resurrection, STEP 0). This work existed only on the
laptop disk.

- chain shadow accrual + probe script (038_chain_shadow.sql)
- WNBA possession feed: ESPN adapter, usage service, verify script
  (039_wnba_player_game.sql)
- baseball chain
- retention/snapshot service updates, tableKeys, matchupKeys
- specs: chain-v1, wnba-possession-feed, wnba-source-survey
- unit tests for the above

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01QnvJAkC3h5QGmb6dipoiWn
This commit is contained in:
Kev
2026-08-14 16:53:37 -04:00
parent 0657b71d18
commit 6c34af3414
22 changed files with 4958 additions and 32 deletions
+377
View File
@@ -0,0 +1,377 @@
'use strict';
/**
* espnWnbaAdapter — the WNBA possession/usage tap.
*
* Spec: specs/wnba-possession-feed.md
*
* ── WHY THIS SOURCE ──────────────────────────────────────────────────────
* The chain's basketball `chainFn` is `usage × possessions × efficiency`, and
* none of those three existed for WNBA anywhere in this repo. The candidates
* were the Python `nba_api` service (WNBA endpoints) and ESPN's free site API.
* The Python service is offline in production, which makes it a source we
* cannot depend on — so this is ESPN, the same host already used for WNBA
* schedules, box scores and live tracking.
*
* ── WHAT THE SOURCE ACTUALLY RETURNS (verified live, 2026-08-13) ─────────
* `summary?event={id}` → `boxscore.players[].statistics[0]` with
*
* keys: minutes, points, fieldGoalsMade-fieldGoalsAttempted,
* threePointFieldGoalsMade-threePointFieldGoalsAttempted,
* freeThrowsMade-freeThrowsAttempted, rebounds, assists, turnovers,
* steals, blocks, offensiveRebounds, defensiveRebounds, fouls, plusMinus
*
* plus `starter` / `didNotPlay` / `ejected` per athlete, and `boxscore.teams[]`
* carrying the team totals (FGA, FTA, totalTurnovers, offensiveRebounds).
*
* ── THE RATES ARE DERIVED, NOT SERVED — AND THAT DISTINCTION IS REAL ─────
* ESPN does NOT return a usage rate, a possession count or a pace figure. It
* returns their COMPONENTS. Usage and possessions are then exact arithmetic
* (the standard box-score identities below), not proxies — every term is a
* counted event, and the only approximation is the 0.44 free-throw-trip
* coefficient, which is the same constant every public implementation uses.
*
* That is worth stating precisely because "derived" and "proxied" are different
* claims: a proxy stands in for a quantity we cannot see, and these are the
* quantity itself, recomputed.
*
* ── POINT-IN-TIME IS STRUCTURAL HERE, NOT RETROFITTED ────────────────────
* `statcast_aggregates` had to grow a `statcast_history` twin because it stores
* a SEASON AGGREGATE upserted in place, destroying every prior version. This
* stores PER GAME rows instead. A completed box score never changes, so any
* as-of profile is `WHERE game_date < asOf` — a filter, not a snapshot table.
* There is nothing to retrofit because there is nothing being overwritten.
*/
const SITE = 'https://site.api.espn.com/apis/site/v2/sports/basketball/wnba';
/** WNBA regulation: 40 minutes, five players on the floor. */
const REGULATION_MINUTES = 40;
const PLAYERS_ON_FLOOR = 5;
/**
* The free-throw-trip coefficient. ~44% of free throws end a possession (the
* rest are the first of a pair). This is the one estimated constant in the
* possession identity and it is the field-standard value.
*/
const FT_TRIP = 0.44;
const num = (v) => {
if (v == null || v === '') return null;
const n = Number(v);
return Number.isFinite(n) ? n : null;
};
/** "8-21" → { made: 8, att: 21 }. A malformed pair is absent, never zero. */
function madeAtt(v) {
const parts = String(v == null ? '' : v).split('-');
if (parts.length !== 2) return { made: null, att: null };
return { made: num(parts[0]), att: num(parts[1]) };
}
/**
* Parse ONE team's player rows out of a summary payload.
*
* A player who did not play is OMITTED, never recorded with zeros. A zero-minute
* row asserts he was available and produced nothing; a DNP is a different fact,
* and the usage denominator would divide by his zero minutes anyway.
*/
function parsePlayers(group) {
const st = group && Array.isArray(group.statistics) ? group.statistics[0] : null;
if (!st || !Array.isArray(st.keys) || !Array.isArray(st.athletes)) return [];
const idx = {};
st.keys.forEach((k, i) => { idx[k] = i; });
const at = (stats, k) => (idx[k] === undefined ? null : stats[idx[k]]);
const out = [];
for (const a of st.athletes) {
if (!a || a.didNotPlay || !Array.isArray(a.stats) || a.stats.length === 0) continue;
const s = a.stats;
const minutes = num(at(s, 'minutes'));
if (minutes === null) continue; // unreadable, not zero
const fg = madeAtt(at(s, 'fieldGoalsMade-fieldGoalsAttempted'));
const fg3 = madeAtt(at(s, 'threePointFieldGoalsMade-threePointFieldGoalsAttempted'));
const ft = madeAtt(at(s, 'freeThrowsMade-freeThrowsAttempted'));
out.push({
source_id: a.athlete && a.athlete.id != null ? String(a.athlete.id) : null,
player_name: (a.athlete && a.athlete.displayName) || null,
starter: a.starter === true,
minutes,
points: num(at(s, 'points')),
fgm: fg.made, fga: fg.att,
fg3m: fg3.made, fg3a: fg3.att,
ftm: ft.made, fta: ft.att,
reb: num(at(s, 'rebounds')),
oreb: num(at(s, 'offensiveRebounds')),
dreb: num(at(s, 'defensiveRebounds')),
ast: num(at(s, 'assists')),
tov: num(at(s, 'turnovers')),
stl: num(at(s, 'steals')),
blk: num(at(s, 'blocks')),
pf: num(at(s, 'fouls')),
plus_minus: num(at(s, 'plusMinus')),
});
}
return out;
}
/** Team totals, from `boxscore.teams[]`. Absent stays absent. */
function parseTeamTotals(teamEntry) {
const m = {};
for (const s of (teamEntry && teamEntry.statistics) || []) m[s.name] = s.displayValue;
const fg = madeAtt(m['fieldGoalsMade-fieldGoalsAttempted']);
const ft = madeAtt(m['freeThrowsMade-freeThrowsAttempted']);
return {
fga: fg.att,
fta: ft.att,
// `totalTurnovers` includes team turnovers (a shot-clock violation belongs
// to the possession count even though no player committed it); the plain
// `turnovers` column does not.
tov: num(m.totalTurnovers) ?? num(m.turnovers),
oreb: num(m.offensiveRebounds),
dreb: num(m.defensiveRebounds),
};
}
/**
* TEAM POSSESSIONS — the standard box-score identity.
*
* POSS = FGA − OREB + TOV + 0.44 × FTA
*
* Any missing term ⇒ null. Treating an absent turnover count as zero would
* understate possessions and inflate every usage rate computed against it.
*/
function possessions(t) {
if (!t) return null;
const { fga, fta, tov, oreb } = t;
if ([fga, fta, tov, oreb].some((v) => v == null)) return null;
return fga - oreb + tov + FT_TRIP * fta;
}
/**
* PACE — possessions per 40 minutes of team play.
*
* `teamMinutes / PLAYERS_ON_FLOOR` is the number of GAME minutes actually
* played, which is how overtime enters without a special case: five players ×
* 45 minutes is 225 team-minutes, and the divisor follows.
*/
function pace(poss, teamMinutes) {
if (poss == null || !(teamMinutes > 0)) return null;
const gameMinutes = teamMinutes / PLAYERS_ON_FLOOR;
if (!(gameMinutes > 0)) return null;
return (poss * REGULATION_MINUTES) / gameMinutes;
}
/**
* USAGE RATE — the share of his team's possessions a player ends while on the
* floor. The standard identity:
*
* USG% = 100 × ((FGA + 0.44·FTA + TOV) × (TmMIN / 5))
* / (MIN × (TmFGA + 0.44·TmFTA + TmTOV))
*
* Null on any missing term or a zero denominator — a usage rate is a ratio, and
* a fabricated one would feed the chain's opportunity term directly.
*/
function usageRate(p, t, teamMinutes) {
if (!p || !t) return null;
if ([p.fga, p.fta, p.tov, p.minutes].some((v) => v == null)) return null;
if ([t.fga, t.fta, t.tov].some((v) => v == null)) return null;
if (!(p.minutes > 0) || !(teamMinutes > 0)) return null;
const denom = p.minutes * (t.fga + FT_TRIP * t.fta + t.tov);
if (!(denom > 0)) return null;
const numer = (p.fga + FT_TRIP * p.fta + p.tov) * (teamMinutes / PLAYERS_ON_FLOOR);
return (100 * numer) / denom;
}
/** TRUE SHOOTING — points per scoring possession. Null without attempts. */
function trueShooting(p) {
if (!p || p.points == null || p.fga == null || p.fta == null) return null;
const tsa = 2 * (p.fga + FT_TRIP * p.fta);
if (!(tsa > 0)) return null;
return p.points / tsa;
}
/** Effective FG% — a three counts for one and a half. */
function efgPct(p) {
if (!p || p.fgm == null || p.fg3m == null || p.fga == null) return null;
if (!(p.fga > 0)) return null;
return (p.fgm + 0.5 * p.fg3m) / p.fga;
}
const round = (v, d) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 10 ** d) / 10 ** d);
/**
* Turn ONE completed game's summary payload into per-player rows.
*
* ONLY COMPLETED GAMES. An in-progress box score is a partial fact whose usage
* denominator is still moving, and storing it would make a row's meaning depend
* on when it was written — the exact property the point-in-time design exists to
* avoid.
*
* @returns {object} { rows, game, skipped }
*/
function parseSummary(payload, opts = {}) {
const comp = payload && payload.header && Array.isArray(payload.header.competitions)
? payload.header.competitions[0] : null;
const status = comp && comp.status && comp.status.type ? comp.status.type.name : null;
const gameId = payload && payload.header ? String(payload.header.id || '') : '';
if (!comp || !gameId) return { rows: [], game: null, skipped: 'no_header' };
if (status !== 'STATUS_FINAL' && !opts.allowUnfinished) {
return { rows: [], game: null, skipped: `not_final:${status || 'unknown'}` };
}
const byId = {};
for (const c of comp.competitors || []) {
if (c && c.team && c.team.id != null) byId[String(c.team.id)] = c;
}
const season = payload.header.season ? Number(payload.header.season.year) : null;
// The ET calendar date of the game. ESPN dates are UTC and a 23:30Z tip is the
// SAME ET evening — using the UTC date would file half the slate a day late,
// which is the `isPreGame` lesson in a different costume.
const gameDate = etDate(comp.date);
const groups = (payload.boxscore && payload.boxscore.players) || [];
const teamTotals = (payload.boxscore && payload.boxscore.teams) || [];
const totalsByTeamId = {};
for (const t of teamTotals) {
if (t && t.team && t.team.id != null) totalsByTeamId[String(t.team.id)] = parseTeamTotals(t);
}
const rows = [];
for (const g of groups) {
const teamId = g && g.team && g.team.id != null ? String(g.team.id) : null;
if (!teamId) continue;
const players = parsePlayers(g);
if (players.length === 0) continue;
const totals = totalsByTeamId[teamId];
const teamMinutes = players.reduce((a, p) => a + (p.minutes || 0), 0);
const poss = possessions(totals);
const me = byId[teamId];
const them = Object.values(byId).find((c) => String(c.team.id) !== teamId);
const teamAbbr = (me && me.team && me.team.abbreviation) || (g.team && g.team.abbreviation) || null;
const oppAbbr = (them && them.team && them.team.abbreviation) || null;
const myScore = me ? num(me.score) : null;
const oppScore = them ? num(them.score) : null;
for (const p of players) {
rows.push({
sport: 'wnba',
season,
game_id: gameId,
game_date: gameDate,
source_id: p.source_id,
player_name: p.player_name,
team: teamAbbr,
opponent: oppAbbr,
home_away: me ? me.homeAway : null,
starter: p.starter,
minutes: p.minutes,
points: p.points,
fgm: p.fgm, fga: p.fga, fg3m: p.fg3m, fg3a: p.fg3a, ftm: p.ftm, fta: p.fta,
reb: p.reb, oreb: p.oreb, dreb: p.dreb,
ast: p.ast, tov: p.tov, stl: p.stl, blk: p.blk, pf: p.pf,
plus_minus: p.plus_minus,
// DERIVED — the three the chainFn is built on.
usage_rate: round(usageRate(p, totals, teamMinutes), 3),
ts_pct: round(trueShooting(p), 4),
efg_pct: round(efgPct(p), 4),
// TEAM CONTEXT, denormalised so a chainFn read needs ONE table. These
// are the possession denominators; recomputing them from a second query
// per row is how a per-prop read becomes a per-prop fetch.
team_minutes: teamMinutes,
team_fga: totals ? totals.fga : null,
team_fta: totals ? totals.fta : null,
team_tov: totals ? totals.tov : null,
team_oreb: totals ? totals.oreb : null,
team_possessions: round(poss, 3),
team_pace: round(pace(poss, teamMinutes), 3),
// GAME STATE — what the `redistribute` hook reads. A blowout moves
// involvement between archetypes, and in basketball that is live rather
// than dormant, so the margin has to be on the row.
team_score: myScore,
opp_score: oppScore,
final_margin: myScore != null && oppScore != null ? myScore - oppScore : null,
});
}
}
return { rows, game: { game_id: gameId, game_date: gameDate, season, status }, skipped: null };
}
/** UTC instant → the ET calendar date it belongs to. */
function etDate(iso) {
if (!iso) return null;
const d = new Date(iso);
if (Number.isNaN(d.getTime())) return null;
return new Intl.DateTimeFormat('en-CA', {
timeZone: 'America/New_York', year: 'numeric', month: '2-digit', day: '2-digit',
}).format(d);
}
const httpJson = async (url, opts = {}) => {
if (typeof opts.fetchImpl === 'function') return opts.fetchImpl(url);
const res = await fetch(url, { headers: { accept: 'application/json' } });
if (!res.ok) throw new Error(`ESPN ${res.status} for ${url}`);
return res.json();
};
/**
* LEAGUE MEMBERSHIP, from the source — never a hardcoded table.
*
* The season pull surfaced two games that are not league games: an exhibition
* against Nigeria (2026-05-02, "NIGER") and the All-Star game (2026-07-25,
* "SPO" vs "COOP"). Their usage and pace context is meaningless for a forward
* projection — an All-Star game has no defence to speak of and an exhibition is
* not the league — so leaving them in would quietly pollute every profile that
* spans those dates.
*
* The filter reads ESPN's own `/teams`, so an expansion franchise is admitted
* the day the league adds it. A hardcoded fifteen would have to be remembered,
* and this league has expanded twice in three years.
*/
let _teamCache = null;
async function getLeagueTeams(opts = {}) {
if (_teamCache && !opts.force) return _teamCache;
const d = await httpJson(`${SITE}/teams`, opts);
const list = ((((d || {}).sports || [])[0] || {}).leagues || [])[0];
const abbrs = ((list && list.teams) || [])
.map((t) => t && t.team && t.team.abbreviation)
.filter(Boolean);
// An EMPTY result is a failed fetch wearing an honest-absence costume (the
// fielding_oaa lesson). Refuse to cache it, and let the caller decide.
if (abbrs.length === 0) return null;
_teamCache = new Set(abbrs);
return _teamCache;
}
/** Completed game ids for one ET date. */
async function getFinalGameIds(dateYYYYMMDD, opts = {}) {
const url = `${SITE}/scoreboard?dates=${String(dateYYYYMMDD).replace(/-/g, '')}`;
const board = await httpJson(url, opts);
return ((board && board.events) || [])
.filter((e) => e && e.status && e.status.type && e.status.type.name === 'STATUS_FINAL')
.map((e) => String(e.id));
}
async function getGameRows(gameId, opts = {}) {
const payload = await httpJson(`${SITE}/summary?event=${encodeURIComponent(gameId)}`, opts);
const parsed = parseSummary(payload, opts);
if (!parsed.rows.length || opts.skipLeagueFilter) return parsed;
const teams = await getLeagueTeams(opts);
if (!teams) return parsed; // unknown membership ⇒ no filtering
const inLeague = parsed.rows.every((r) => teams.has(r.team) && teams.has(r.opponent));
if (inLeague) return parsed;
const bad = [...new Set(parsed.rows.flatMap((r) => [r.team, r.opponent]))]
.filter((t) => !teams.has(t));
return { rows: [], game: parsed.game, skipped: `not_a_league_game:${bad.join(',')}` };
}
module.exports = {
parseSummary, parsePlayers, parseTeamTotals,
possessions, pace, usageRate, trueShooting, efgPct, madeAtt, etDate,
getFinalGameIds, getGameRows, getLeagueTeams,
SITE, FT_TRIP, REGULATION_MINUTES, PLAYERS_ON_FLOOR,
};
+272
View File
@@ -0,0 +1,272 @@
'use strict';
/**
* baseballChain — BASEBALL'S chainFn. The fixed-opportunity case.
*
* `chain.js` describes a slot: `chainFn(atom, context) → per-entity
* probability`. This fills it for MLB, and it is deliberately the SIMPLE end of
* the problem, because the point of the first fill is to prove the slot rather
* than to win an argument about basketball.
*
* ── WHY BASEBALL IS THE CLEAN CASE ───────────────────────────────────────
* A chained forecast is always `rate × opportunity`. In basketball the
* opportunity term is itself a contested model — possessions are finite and
* teammates compete for them, so one player's usage is another's non-usage, and
* the opportunity term moves with game script. In baseball it does not. The
* batting order is fixed before first pitch, a nine-run lead does not change who
* bats next, and plate appearances per game vary over a narrow, stable range set
* almost entirely by lineup slot.
*
* So here opportunity is a STABLE MULTIPLIER, not a model:
*
* p_hit_per_PA ← skillProjection.paOutcome (the rate — the modelled part)
* expected PA ← lineup slot (the opportunity — a lookup)
* P(hits ≥ k) ← Binomial(PA, p) mixed over the PA distribution
*
* ── THE ATOMS THIS ROUTES ────────────────────────────────────────────────
* Everything below already existed and was reachable only from `scripts/`:
*
* skillProjection.fromStatcastRow — the ONE legal units conversion. Statcast
* stores PERCENTAGES (0–100); feeding raw
* rows in makes `bip = 1 − k − bb` negative
* and refuses almost every row.
* skillProjection.paOutcome — the PA outcome tree: K and BB via log5
* odds-ratio against league, the remainder
* balls in play, and hit-on-contact from
* the ARCHETYPE-SELECTED skill inputs
* (barrel / hard-hit / exit velo / GB-speed)
* against the pitcher's contact allowed and
* the park.
* skillProjection.paDistribution — PA is not fixed at an integer; a hitter
* skillProjection.binomialPmf gets 4 or 5 depending on how the lineup
* skillProjection.atLeast turns over, so the projection MIXES rather
* than pretending PA is known.
*
* Nothing new is modelled here. This is a router, and saying so is the point:
* the reason the "portable core" was not portable is that this routing did not
* exist, so the sport-specific work lived in scripts that no pipeline called.
*
* ── REFUSAL, NOT SUBSTITUTION ────────────────────────────────────────────
* No batter profile ⇒ null ⇒ `chain.applyChainFn` DROPS the atom. It never
* becomes p=0 (which would zero a whole ticket) and never falls back to a league
* hitter (which would assert we had read someone we had not). A missing pitcher
* is different and is handled inside `paOutcome`: the batter's own rate stands
* untouched rather than being pulled toward league.
*
* ── SCOPE: HITS ONLY, ON PURPOSE ─────────────────────────────────────────
* `projectSkill` also routes total_bases, but TB's head-to-head is on record as
* INCONCLUSIVE-under-contamination, and the point-in-time window that would make
* it adjudicable only starts once `statcast_history` has accrued. Widening the
* first shadow to a stat whose verdict is already muddy would buy noise. Any
* other stat returns null and its atom is dropped.
*/
const sk = require('./skillProjection');
const pss = require('./platoonSeverity');
const { knownNumber, knownRate } = require('../../utils/known');
/** Only this stat is chained today — see the header. */
const CHAINED_STATS = Object.freeze(['hits']);
/**
* EXPECTED PLATE APPEARANCES BY LINEUP SLOT — the opportunity term.
*
* This is a LOOKUP, not a model, and that is the whole claim being made about
* baseball: the leadoff hitter gets roughly three quarters of a plate appearance
* more per game than the nine hole, the decline is close to linear at about a
* tenth of a PA per slot, and it does not respond to game state. The numbers
* below are that documented structure, not a fit — nothing here was tuned on
* settled rows, because a tuned opportunity term on 1,741 rows is curve-fitting
* dressed as physics.
*
* An unknown slot falls to `skillProjection.DEFAULT_PA` (4.1, "a regular"),
* which is a statement about a typical starter and is applied ONLY when the
* lineup has not posted. It is the one substitution in this module and it is
* confined to the opportunity term — never to the rate.
*/
const PA_BY_SLOT = Object.freeze({
1: 4.65, 2: 4.55, 3: 4.45, 4: 4.35, 5: 4.25, 6: 4.15, 7: 4.05, 8: 3.95, 9: 3.85,
});
function expectedPaForSlot(slot) {
const s = knownNumber(slot);
if (s === null) return null; // absent slot ⇒ caller's default
const k = Math.round(s);
return PA_BY_SLOT[k] ?? null;
}
/**
* Normalize a statcast row into the profile `paOutcome` expects.
*
* `fromStatcastRow` is the single legal chokepoint for the percentage→fraction
* conversion; calling it here rather than at each call site is deliberate.
*/
function profileFrom(row) {
if (!row) return null;
try { return sk.fromStatcastRow(row) || null; } catch { return null; }
}
/**
* P(stat ≥ line) for ONE hitter, chained from the PA rate over expected PA.
*
* @param {object} atom
* statType 'hits'
* line the market line (0.5 ⇒ target 1)
* batterRow raw `statcast_aggregates` row for the hitter
* pitcherRow raw row for the opposing starter (absent ⇒ batter's own rate)
* archetype selects WHICH skill inputs drive this hitter
* park park multiplier (absent ⇒ 1, i.e. no claim)
* lineupSlot batting order position — the opportunity term
* expectedPa an explicit override, used ahead of the slot lookup
* @param {object} context slate-level context; `allowed` gates which features
* may be read (the shadow runs on CANDIDATEs)
* @returns {object|null} `{ p, ... }` merged onto the atom, or null ⇒ DROPPED
*/
/**
* THE HAND SPLIT — the matchup conditioner, applied to the RATE.
*
* `paOutcome` reads a hitter's SEASON rates. Those already contain his platoon
* split, averaged over whichever hands he happened to face — which is precisely
* the thing a forward matchup read is supposed to un-average. `platoonRead`
* gives the severity of THIS hitter's own split (shrunk by the smaller side's
* sample, refused outright below 60 PA) in the direction tonight's matchup
* actually runs.
*
* IT ENTERS AT THE PER-PA HIT RATE, NOT AT THE OUTPUT PROBABILITY. Multiplying
* P(hits ≥ 1) by a platoon factor would be a different and wrong claim: the
* split is a statement about how often a plate appearance becomes a hit, and the
* binomial over plate appearances is what turns that into a threshold
* probability. Applying it after the chain would scale a number that has already
* been through the opportunity term.
*
* UNREADABLE ⇒ NO ADJUSTMENT, WITH A REASON. The season rate stands untouched —
* never nudged toward a league-typical split, which is the claim
* `platoonSeverity` refuses to make on a thin sample.
*/
function handSplitFor(atom) {
if (!atom || !atom.platoonSplits) return { multiplier: 1, applied: false, reason: 'no_splits' };
if (!atom.throws) return { multiplier: 1, applied: false, reason: 'no_pitcher_hand' };
if (!atom.bats) return { multiplier: 1, applied: false, reason: 'no_batter_hand' };
const read = pss.platoonRead({ splits: atom.platoonSplits, bats: atom.bats, throws: atom.throws });
if (!read) return { multiplier: 1, applied: false, reason: 'unreadable' };
if (!read.readable || !Number.isFinite(read.multiplier)) {
return { multiplier: 1, applied: false, reason: read.reason || 'unreadable' };
}
return {
multiplier: read.multiplier,
applied: true,
reason: null,
observed_split: read.observed_split,
smaller_side_pa: read.smaller_side_pa,
facing_opposite_hand: read.facing_opposite_hand,
};
}
/**
* P(stat ≥ line) for ONE hitter, chained from the PA rate over expected PA.
*
* The chain is written out here rather than delegated to `projectSkill` because
* the hand split has to enter BETWEEN the rate and the opportunity term, and
* `projectSkill` composes those two in one call. With no split supplied this is
* arithmetically identical to `projectSkill` — a test asserts that, so the
* restructuring cannot quietly become a second model.
*
* @param {object} atom
* statType/line/batterRow/pitcherRow/archetype/park/lineupSlot/expectedPa
* bats / throws / platoonSplits — the hand split (A4-dated by the caller)
* @param {object} context
* allowed which features may be read (the shadow runs on CANDIDATEs)
* onRefusal(id, reason) measurement side-channel; refusals are COUNTED, and
* a refusal always DROPS the atom — there is no
* fallback to a season rate anywhere in this module,
* because a silent fallback is indistinguishable from
* a read and would make the shadow's agreement with the
* counter meaningless.
* @returns {object|null} `{ p, ... }` merged onto the atom, or null ⇒ DROPPED
*/
function chainFn(atom, context = {}) {
const refuse = (reason) => {
if (typeof context.onRefusal === 'function') {
try { context.onRefusal(atom && atom.id, reason); } catch { /* measurement never breaks a read */ }
}
return null;
};
if (!atom) return refuse('no_atom');
const stat = String(atom.statType || atom.stat_type || '').toLowerCase();
if (!CHAINED_STATS.includes(stat)) return refuse('stat_not_chained');
const line = knownNumber(atom.line);
if (line === null) return refuse('no_line');
const batter = profileFrom(atom.batterRow);
if (!batter) return refuse('no_batter_profile'); // absent, never a league hitter
const pitcher = profileFrom(atom.pitcherRow); // null is fine — paOutcome copes
const park = knownRate(atom.park) ?? 1;
const allowed = context.allowed || null;
const archetype = atom.archetype || null;
// ── 1. THE RATE ────────────────────────────────────────────────────────
const pa = sk.paOutcome({ batter, pitcher, park, archetype, allowed });
if (!pa || !Number.isFinite(pa.p_hit_per_pa)) return refuse('pa_outcome_refused');
// ── 2. THE MATCHUP CONDITIONER ─────────────────────────────────────────
const split = handSplitFor(atom);
const pHit = Math.min(1, Math.max(0, pa.p_hit_per_pa * split.multiplier));
// ── 3. THE OPPORTUNITY TERM ────────────────────────────────────────────
const expectedPa = knownNumber(atom.expectedPa)
?? expectedPaForSlot(atom.lineupSlot)
?? null; // null ⇒ paDistribution's DEFAULT_PA
const paPmf = sk.paDistribution(expectedPa);
// ── 4. THE CHAIN ───────────────────────────────────────────────────────
const pmf = new Array(sk.PA_CAP + 1).fill(0);
for (let n = 0; n < paPmf.length; n += 1) {
if (!paPmf[n]) continue;
const bp = sk.binomialPmf(n, pHit);
for (let x = 0; x < bp.length; x += 1) pmf[x] += paPmf[n] * bp[x];
}
const target = Math.max(1, Math.ceil(line));
const p = sk.atLeast(pmf, target);
if (!Number.isFinite(p)) return refuse('chain_produced_no_probability');
const r3 = (v) => (Number.isFinite(v) ? Math.round(v * 1000) / 1000 : null);
return {
p,
chain_projected_value: r3(pmf.reduce((a, q, i) => a + q * i, 0)),
chain_per_pa: {
k_rate: r3(pa.k_rate), bb_rate: r3(pa.bb_rate), bip_rate: r3(pa.bip_rate),
hit_on_contact: r3(pa.hit_on_contact),
p_hit_per_pa_season: r3(pa.p_hit_per_pa),
p_hit_per_pa: r3(pHit),
},
chain_expected_pa: expectedPa,
chain_opportunity_source: knownNumber(atom.expectedPa) !== null
? 'explicit'
: (expectedPaForSlot(atom.lineupSlot) !== null ? 'lineup_slot' : 'default_regular'),
chain_pitcher_applied: pa.inputs_used.pitcher_applied,
// The hand split, recorded whether or not it fired. A non-application is a
// fact about the sample, and hiding it would make the fire rate a fiction.
chain_platoon_applied: split.applied,
chain_platoon_multiplier: split.applied ? split.multiplier : null,
chain_platoon_reason: split.reason,
chain_platoon_observed_split: split.observed_split ?? null,
chain_platoon_smaller_side_pa: split.smaller_side_pa ?? null,
chain_archetype: String(archetype || 'DEFAULT').toUpperCase(),
chain_family: 'pa_outcome_tree_binomial',
};
}
/** The hook `chain.chainUp`/`chainAcross` take. Dormant in baseball — see chain.js. */
function redistribute(legs) {
// A nine-run lead does not change who bats next. Returning the legs UNCHANGED
// is the honest dormant behaviour; returning null would read as "the hook
// failed" rather than "this sport has no redistribution".
return legs;
}
module.exports = {
chainFn, expectedPaForSlot, redistribute, handSplitFor,
PA_BY_SLOT, CHAINED_STATS,
};
+123 -24
View File
@@ -29,7 +29,14 @@
* the bench). DORMANT in baseball — a nine-run lead does
* not change who bats next — and LIVE in basketball,
* where it is most of the edge. The hook exists here so
* basketball is content rather than a rewrite.
* basketball is content rather than a rewrite. It runs
* on BOTH readings: a redistribution that reached only
* the team read would make `selfCheck` flag an
* inconsistency the model had itself just created.
*
* All five are real parameters. `chainFn` was described here and absent from the
* code for its whole life, which is why every sport's actual work happened
* outside the "portable" core — see `applyChainFn` below.
*
* ── CALIBRATION IS A HARD PRECONDITION, NOT A WARNING ────────────────────
* Errors that are survivable one at a time MULTIPLY when chained. Measured on
@@ -55,19 +62,89 @@ function usableAtoms(atoms) {
return (atoms || []).filter((a) => a && knownNumber(a.p) !== null);
}
/**
* THE chainFn SLOT — atom + context → per-entity probability.
*
* This is the stage the header always described and the code never had. Without
* it, every atom had to arrive with `p` already computed somewhere else, which
* is exactly how the sport-specific work ended up living outside the machine
* that claims to be portable. Baseball fills it with a fixed-opportunity read
* (p per plate appearance, chained over expected PA); basketball will fill it
* with a contested-possession read. Neither is a code path here.
*
* DEFAULT IS IDENTITY-ON-p, so every caller that passes a pre-computed
* probability keeps working unchanged.
*
* A chainFn that returns null — or throws — makes that atom UNREADABLE, which
* means DROPPED, never p=0. A zero leg would zero an entire ticket, and "we
* could not read him" is not "he cannot do it". The refusals are COUNTED in the
* result rather than swallowed, so a chainFn that is quietly failing on the
* whole board is visible instead of looking like a thin slate.
*/
function applyChainFn(atoms, opts = {}) {
const fn = typeof opts.chainFn === 'function' ? opts.chainFn : null;
if (!fn) return { atoms: atoms || [], applied: false, refused: 0 };
const ctx = opts.context || {};
let refused = 0;
const out = [];
for (const a of atoms || []) {
if (!a) { refused += 1; continue; }
let r;
try { r = fn(a, ctx); } catch { r = null; }
if (r === null || r === undefined) { refused += 1; continue; }
const merged = typeof r === 'object' ? { ...a, ...r } : { ...a, p: r };
if (knownNumber(merged.p) === null) { refused += 1; continue; }
out.push(merged);
}
return { atoms: out, applied: true, refused };
}
/** The archetype-redistribution hook, applied identically by BOTH readings. */
function applyRedistribute(legs, opts = {}) {
if (typeof opts.redistribute !== 'function') return { legs, redistributed: false };
const out = opts.redistribute(legs, opts.context || {});
if (!Array.isArray(out) || out.length === 0) return { legs, redistributed: false };
return { legs: usableAtoms(out), redistributed: true };
}
/**
* PREPARE — chainFn, then usability, then redistribution, in that order.
*
* Exported so a caller can prepare ONCE and hand the SAME legs to both readings.
* That matters: `selfCheck` compares the across-read to the up-read, so if the
* two were prepared separately and only one saw a redistribution, the check
* would flag an inconsistency that the model itself had just created.
*/
function prepareAtoms(atoms, opts = {}) {
const c = applyChainFn(atoms, opts);
const usable = usableAtoms(c.atoms);
const r = applyRedistribute(usable, opts);
return {
legs: r.legs,
chain_fn_applied: c.applied,
chain_fn_refused: c.refused,
redistributed: r.redistributed,
};
}
/**
* CHAIN ACROSS — compound probability of every leg landing.
*
* @param {Array} atoms [{ id, p, calibrated, gameId, entityId, ... }]
* @param {object} opts
* correlation(a, b) → 0..1 shared-variance estimate between two legs
* chainFn(atom, context) → per-entity probability (default: identity on `p`)
* correlation(a, b) → -1..1 dependence estimate between two legs
* redistribute(legs, context) → reweighted legs (same hook chainUp has)
* requireCalibrated (default TRUE) — see the header
* @returns {object|null} refusal is explicit and reasoned, never a silent 0
*/
function chainAcross(atoms, opts = {}) {
const requireCalibrated = opts.requireCalibrated !== false;
const legs = usableAtoms(atoms);
if (legs.length === 0) return { ok: false, reason: 'no_usable_atoms' };
const prep = prepareAtoms(atoms, opts);
const legs = prep.legs;
if (legs.length === 0) {
return { ok: false, reason: 'no_usable_atoms', chain_fn_refused: prep.chain_fn_refused };
}
if (requireCalibrated) {
const uncal = legs.filter((l) => l.calibrated !== true);
@@ -82,10 +159,16 @@ function chainAcross(atoms, opts = {}) {
}
}
// Independent product first, then a correlation discount. Same-game legs share
// the pitcher, the park and the weather, so treating them as independent
// OVERSTATES the ticket — the error runs in the flattering direction, which is
// exactly the one to be careful about.
// Independent product first, then a correlation adjustment.
//
// ── THE SIGN IS NOT OPTIONAL ─────────────────────────────────────────────
// Correlation used to be clamped to [0, 1], which silently asserted that legs
// can only ever land TOGETHER. That is baseball's shape — same-game legs share
// the pitcher, the park and the weather — and it is the wrong shape for a
// sport where entities compete for the same finite opportunity. Two teammates'
// scoring props are negatively correlated through shared possessions: one
// player's shot is another player's non-shot. Under the old clamp that case
// was not merely mismodelled, it was INEXPRESSIBLE.
const corrFn = typeof opts.correlation === 'function' ? opts.correlation : () => 0;
let independent = 1;
for (const l of legs) independent *= Math.min(1, Math.max(0, Number(l.p)));
@@ -95,16 +178,26 @@ function chainAcross(atoms, opts = {}) {
for (let i = 0; i < legs.length; i += 1) {
for (let j = i + 1; j < legs.length; j += 1) {
const c = Number(corrFn(legs[i], legs[j]));
if (Number.isFinite(c)) { corrSum += Math.min(1, Math.max(0, c)); pairs += 1; }
if (Number.isFinite(c)) { corrSum += Math.min(1, Math.max(-1, c)); pairs += 1; }
}
}
const meanCorr = pairs > 0 ? corrSum / pairs : 0;
// Positive correlation makes the JOINT more likely than independence implies
// (legs tend to land together), so the adjustment moves toward the weakest leg
// — bounded, and stated as an approximation rather than a derivation.
const weakest = Math.min(...legs.map((l) => Number(l.p)));
const compound = independent + meanCorr * (weakest - independent);
// The two targets are the FRÉCHET–HOEFFDING bounds on a joint, which is what
// makes the direction principled rather than chosen:
//
// perfectly co-monotone (corr = +1) → joint = min(p_i) — the upper bound
// independent (corr = 0) → joint = Π p_i
// perfectly counter-monotone (corr = −1) → joint = max(0, Σp − (n−1)) — the lower bound
//
// So |corr| interpolates from independence toward the bound its SIGN selects.
// Positive is unchanged from before (the weakest leg IS the upper bound), so
// every previously-correct number stays exactly what it was.
const probs = legs.map((l) => Math.min(1, Math.max(0, Number(l.p))));
const frechetUpper = Math.min(...probs);
const frechetLower = Math.max(0, probs.reduce((a, b) => a + b, 0) - (probs.length - 1));
const target = meanCorr >= 0 ? frechetUpper : frechetLower;
const compound = independent + Math.abs(meanCorr) * (target - independent);
return {
ok: true,
@@ -112,8 +205,12 @@ function chainAcross(atoms, opts = {}) {
independent_probability: round4(independent),
mean_pairwise_correlation: round4(meanCorr),
compound_probability: round4(Math.min(1, Math.max(0, compound))),
correlation_direction: meanCorr > 0 ? 'toward_joint' : meanCorr < 0 ? 'apart' : 'independent',
cross_game_legs: new Set(legs.map((l) => l.gameId)).size,
correlation_caveat: 'pairwise mean, applied as a bounded shift toward the weakest leg — an approximation, not a joint distribution',
chain_fn_applied: prep.chain_fn_applied,
chain_fn_refused: prep.chain_fn_refused,
redistributed: prep.redistributed,
correlation_caveat: 'pairwise mean interpolated toward the Fréchet bound its sign selects — an approximation, not a joint distribution',
};
}
@@ -124,13 +221,10 @@ function chainAcross(atoms, opts = {}) {
* and may return a reweighted set — dormant in baseball, live in basketball.
*/
function chainUp(atoms, opts = {}) {
let legs = usableAtoms(atoms);
if (legs.length === 0) return { ok: false, reason: 'no_usable_atoms' };
let redistributed = false;
if (typeof opts.redistribute === 'function') {
const out = opts.redistribute(legs, opts.context || {});
if (Array.isArray(out) && out.length > 0) { legs = usableAtoms(out); redistributed = true; }
const prep = prepareAtoms(atoms, opts);
const legs = prep.legs;
if (legs.length === 0) {
return { ok: false, reason: 'no_usable_atoms', chain_fn_refused: prep.chain_fn_refused };
}
const weightOf = (a) => {
@@ -142,7 +236,9 @@ function chainUp(atoms, opts = {}) {
ok: true,
contributors: legs.length,
expected_value: round4(expected),
redistributed,
chain_fn_applied: prep.chain_fn_applied,
chain_fn_refused: prep.chain_fn_refused,
redistributed: prep.redistributed,
};
}
@@ -223,4 +319,7 @@ function propagate(atom, observation, opts = {}) {
const round4 = (v) => (Number.isFinite(v) ? Math.round(v * 10000) / 10000 : v);
module.exports = { chainAcross, chainUp, selfCheck, propagate, usableAtoms };
module.exports = {
chainAcross, chainUp, selfCheck, propagate,
usableAtoms, prepareAtoms, applyChainFn, applyRedistribute,
};
+359
View File
@@ -0,0 +1,359 @@
'use strict';
/**
* chainShadow — run the chain on the live board and SERVE NONE OF IT.
*
* This is the A6 shape, one layer up. A6 proved the hits factors could fire,
* froze what they WOULD have applied, and changed no served number. This does
* the same for the whole forecast: the chain computes a probability for every
* MLB hits prop on the slate, it is written to its own column beside
* `factor_inputs`, and the counter continues to serve untouched.
*
* ── WHAT IS RECORDED, AND WHY IT IS THAT SHAPE ───────────────────────────
* The unit of evidence is the TRIPLE:
*
* (chain_p, counter_p, outcome)
*
* All three on one row, side-aligned. `counter_p` is the served
* `estimateProbability` value for that exact side, `outcome` arrives later from
* the ordinary settle pass, and `chain_p` is this. Evidence you cannot
* adjudicate is not evidence: a chain probability stored without the number it
* must beat, or without the result, can only ever be compared to itself — which
* is the same defect `factorFreeze` exists to avoid one level down.
*
* ── UN-SERVABLE, AND SAID SO IN THE DATA ─────────────────────────────────
* `chainAcross` is called with `requireCalibrated: false`, which is legitimate
* ONLY because nothing downstream reads the result. The chain's probabilities
* have never been through a calibration gate, and compounding an uncalibrated
* probability is the single most harmful thing this product could ship. So every
* row carries `servable: false` and `status: 'UN-SERVABLE'` explicitly, in the
* stored payload rather than only in a comment. Flipping that is a separate,
* gated event and it requires passing the gate, not editing this file.
*
* ── THE SELF-CHECK IS VACUOUS TODAY, AND THAT IS REPORTED, NOT HIDDEN ────
* `selfCheck` earns its keep by comparing the per-entity reads to an INDEPENDENT
* team read. No independent team read exists — the game-script projection was
* deliberately not built, because no atom has passed the gate and building it
* would be plausibility rather than proof. So the up-read here is assembled from
* the SAME atoms as the across-read, and agreement between them is arithmetic,
* not evidence. The result is labelled `vacuous: true` so nobody later mistakes
* a tautology for a passing consistency test.
*/
const chain = require('./chain');
const baseball = require('./baseballChain');
const { nameKey } = require('../../utils/playerName');
const { knownNumber } = require('../../utils/known');
const SHADOW_VERSION = 'chain-shadow@1';
/** Only MLB hits are chained today — see baseballChain's header. */
const SHADOW_STAT = 'hits';
const r4 = (v) => (Number.isFinite(v) ? Math.round(v * 10000) / 10000 : null);
/** `player_key|stat|line` — direction-free, because the chain reads P(over). */
function shadowKey(playerKey, stat, line) {
const l = knownNumber(line);
return `${playerKey}|${String(stat || '').toLowerCase()}|${l === null ? '' : l}`;
}
/**
* The UP reading's chainFn: the same atoms, asked for an expected COUNT rather
* than a threshold probability. This is the pluggable slot doing exactly the job
* it exists for — one atom set, two questions, no second engine.
*/
function upChainFn(atom, context) {
const across = baseball.chainFn(atom, context);
if (!across) return null;
const v = knownNumber(across.chain_projected_value);
return v === null ? null : { p: v };
}
/** The same question, read off legs the chain has ALREADY computed. */
function upFromPrepared(leg) {
const v = knownNumber(leg && leg.chain_projected_value);
return v === null ? null : { p: v };
}
/**
* Build one atom per graded MLB hits prop.
*
* Every input is READ from what the snapshot already loaded — the statcast rows
* the challenger fetched, the archetype the enrichment resolved, the park factor
* arch-v1 attached. No new I/O: this runs inside a cron slot that already grades
* the whole board, and adding a fetch per prop is how a measurement layer turns
* into an outage.
*/
function buildAtoms(grades, deps = {}) {
const statcastByKey = deps.statcastByKey || null;
const pitcherRowFor = typeof deps.pitcherRowFor === 'function' ? deps.pitcherRowFor : () => null;
const lineupSlotFor = typeof deps.lineupSlotFor === 'function' ? deps.lineupSlotFor : () => null;
// THE HAND SPLIT, resolved AS-OF-CORRECT by the caller (A4 discipline: the
// split as it was known at grade time, or nothing). This layer never dates
// anything itself — it takes what it is handed, so there is exactly one place
// the as-of rule lives and it is the place that reads the tables.
const handSplitFor = typeof deps.handSplitFor === 'function' ? deps.handSplitFor : () => null;
const atoms = [];
for (const g of grades || []) {
const stat = String(g.stat_type || g.stat || '').toLowerCase();
if (stat !== SHADOW_STAT) continue;
const player = g.player || g.player_name;
if (!player) continue;
const pk = nameKey(player);
// The LOCKED line is the one the grade was made against; falling back to the
// current line would score the chain on a number the counter never saw.
const line = knownNumber(g.gradedAt && g.gradedAt.line) ?? knownNumber(g.line);
if (line === null) continue;
atoms.push({
id: shadowKey(pk, stat, line),
entityId: pk,
player,
player_key: pk,
gameId: g.game_id || null,
team: g.team || null,
statType: stat,
line,
side: String(g.direction || '').toLowerCase() || null,
batterRow: statcastByKey ? statcastByKey.get(pk) || null : null,
pitcherRow: pitcherRowFor(g),
archetype: g.archetype || null,
park: knownNumber(g.env_park_base),
lineupSlot: lineupSlotFor(g),
// bats/throws/platoonSplits — absent stays absent; `baseballChain` then
// leaves the season rate untouched and records WHY, rather than nudging
// toward a league-typical split.
...(() => {
// A failing split resolver degrades to a SEASON read — it must never
// take the slate down, and it must never be mistaken for a matchup read
// either, which is why the reason is recorded downstream rather than
// swallowed here.
let hs = null;
try { hs = handSplitFor(g); } catch { hs = null; }
hs = hs || {};
return {
bats: hs.bats || null,
throws: hs.throws || null,
platoonSplits: hs.platoonSplits || null,
};
})(),
// The counter's number for THIS row, carried so the triple is assembled
// from what was actually served rather than recomputed later.
counter_p_win: knownNumber(g.p_win),
counter_side: String(g.direction || '').toLowerCase() || null,
// NEVER true here. The chain has not been through a calibration gate.
calibrated: false,
});
}
return atoms;
}
/**
* Run the shadow over one slate.
*
* @returns {object} { version, byKey, summary } — `byKey` maps the direction-free
* shadow key to the per-prop block; the merge into retention rows side-aligns.
*/
function runShadow(grades, deps = {}) {
const atoms = buildAtoms(grades, deps);
// REFUSALS ARE COUNTED BY REASON, not just totalled. A bare count cannot tell
// "the board has no statcast profiles" from "the PA tree has nothing to read",
// and those need opposite fixes. There is NO fallback path: a refused atom is
// dropped, never quietly replaced by a season rate — a silent fallback would
// make the shadow agree with the counter for a reason that looks like
// agreement and is not.
const refusalReasons = new Map();
const bump = (m, k) => m.set(k, (m.get(k) || 0) + 1);
const context = {
allowed: deps.allowed || null,
onRefusal: (_id, reason) => bump(refusalReasons, reason || 'unknown'),
...(deps.context || {}),
};
// ACROSS, per game. A ticket is built from one game's legs far more often than
// from the whole board, and the correlation question only means anything
// within a game. `requireCalibrated:false` — see the header.
const byGame = new Map();
for (const a of atoms) {
const k = a.gameId || 'unbound';
if (!byGame.has(k)) byGame.set(k, []);
byGame.get(k).push(a);
}
const perProp = new Map();
const gameReads = [];
const platoonReasons = new Map();
const opportunityReasons = new Map();
let readable = 0;
let refused = 0;
let platoonApplied = 0;
let opportunityPosted = 0;
for (const [gameId, gameAtoms] of byGame) {
// PREPARED ONCE, and the two readings run on the PREPARED legs.
//
// Two reasons, and the first is a correctness bug avoided: chainAcross,
// chainUp and prepareAtoms each apply the chainFn, so running the raw atoms
// through all three would evaluate every hitter three times and count every
// refusal three times — a fire rate off by 3x, in the flattering direction
// for the refusal count. The second is that identical legs is precisely what
// makes the self-check meaningful at all.
const prep = chain.prepareAtoms(gameAtoms, {
chainFn: baseball.chainFn, redistribute: baseball.redistribute, context,
});
// Identity-on-p: the legs already carry the chained probability.
const across = chain.chainAcross(prep.legs, {
requireCalibrated: false,
correlation: typeof deps.correlation === 'function' ? deps.correlation : undefined,
});
// The UP reading asks the same atoms for an expected COUNT — read off the
// value the chain already produced, not recomputed.
const up = chain.chainUp(prep.legs, { chainFn: upFromPrepared });
const check = chain.selfCheck({
perEntity: prep.legs,
teamRead: up.ok ? up.expected_value : null,
});
refused += knownNumber(prep.chain_fn_refused) ?? 0;
readable += prep.legs.length;
gameReads.push({
game_id: gameId,
legs: prep.legs.length,
refused: knownNumber(across.chain_fn_refused) ?? 0,
compound_probability: across.ok ? across.compound_probability : null,
independent_probability: across.ok ? across.independent_probability : null,
expected_team_hits: up.ok ? up.expected_value : null,
self_check_confidence: check.confidence,
});
const gameBlock = {
game_id: gameId,
across_compound: across.ok ? across.compound_probability : null,
across_legs: across.ok ? across.legs : 0,
up_expected_hits: up.ok ? up.expected_value : null,
self_check: {
confidence: check.confidence,
internal_divergence: check.internal_divergence,
// Both readings derive from the same atoms today, so agreement is
// arithmetic. Labelled rather than quietly reported as a pass.
vacuous: true,
vacuous_reason: 'no independent team read exists — the up-read is assembled from the same atoms as the across-read',
},
};
for (const leg of prep.legs) {
const chainP = knownNumber(leg.p);
if (chainP === null) continue;
perProp.set(leg.id, {
v: SHADOW_VERSION,
status: 'UN-SERVABLE',
servable: false,
servable_reason: 'the chain has not passed a calibration gate; requireCalibrated was false for this read',
stat: leg.statType,
line: leg.line,
// The chain's own quantity is P(over the line); the merge side-aligns it.
chain_p_over: r4(chainP),
chain_projected_value: r4(knownNumber(leg.chain_projected_value)),
chain_per_pa: leg.chain_per_pa || null,
expected_pa: knownNumber(leg.chain_expected_pa),
opportunity_source: leg.chain_opportunity_source || null,
pitcher_applied: leg.chain_pitcher_applied === true,
// THE HAND SPLIT, recorded either way. Whether the matchup conditioner
// fired is the difference between a forward read and a season read, and
// a later adjudication that could not tell them apart would be pooling
// two different models.
platoon_applied: leg.chain_platoon_applied === true,
platoon_multiplier: knownNumber(leg.chain_platoon_multiplier),
platoon_reason: leg.chain_platoon_reason || null,
platoon_observed_split: knownNumber(leg.chain_platoon_observed_split),
platoon_smaller_side_pa: knownNumber(leg.chain_platoon_smaller_side_pa),
archetype: leg.chain_archetype || null,
family: leg.chain_family || null,
game: gameBlock,
});
}
}
// COUNTED OFF THE STORED BLOCKS, not off the legs.
//
// A prop with both an over and an under row produces TWO legs with the same
// id, and `perProp` collapses them into one block. Counting the hand split per
// LEG while the divergence is later sliced per BLOCK gives two rates over two
// different denominators that look comparable and are not — the first draft
// reported 48.1% and 55.1% for the same fact. One denominator, taken from the
// thing that is actually persisted.
for (const block of perProp.values()) {
if (block.platoon_applied === true) platoonApplied += 1;
else bump(platoonReasons, block.platoon_reason || 'unknown');
// A POSTED lineup slot is a real opportunity term; `default_regular` is the
// league's typical starter asserted about this hitter. Counted apart,
// because a chain running on a default opportunity term is doing half the
// job it claims and must not report as if it read the lineup.
if (block.opportunity_source === 'lineup_slot' || block.opportunity_source === 'explicit') {
opportunityPosted += 1;
} else bump(opportunityReasons, block.opportunity_source || 'unknown');
}
return {
version: SHADOW_VERSION,
byKey: perProp,
summary: {
atoms: atoms.length,
readable,
refused,
// Unique props (a prop's over and under legs share one block). This is the
// denominator `platoon_applied` is over.
props: perProp.size,
// The two rates that matter, and they are DIFFERENT questions:
// readable — did the chain produce a probability at all?
// platoon — did it read tonight's MATCHUP, or this season's average?
// Conflating them is how a season-rate read gets reported as a forward one.
platoon_applied: platoonApplied,
// The THIRD distinct question: did the chain read a POSTED lineup slot, or
// fall to "a regular"? rate x opportunity — reporting only the rate half
// as a fire rate is how v2 shipped with a constant 4.1 PA for every hitter.
opportunity_posted: opportunityPosted,
refusal_reasons: Object.fromEntries(refusalReasons),
platoon_reasons: Object.fromEntries(platoonReasons),
opportunity_reasons: Object.fromEntries(opportunityReasons),
games: byGame.size,
games_read: gameReads.filter((g) => g.compound_probability != null).length,
per_game: gameReads,
},
};
}
/**
* Side-align the chain probability onto ONE settled-or-pending row.
*
* `p_win` is expressed for the graded side, so an under row must carry
* `1 − p_over` or the two halves of the triple would be pointing in opposite
* directions and every comparison after it would be inverted. This is the same
* trap `snapshotSettlementService` documents for `outcome` vs `actual_value`.
*/
function alignToSide(block, side, counterP) {
if (!block) return null;
const s = String(side || '').toLowerCase();
const pOver = knownNumber(block.chain_p_over);
if (pOver === null || (s !== 'over' && s !== 'under')) return null;
const chainP = s === 'under' ? 1 - pOver : pOver;
const counter = knownNumber(counterP);
return {
...block,
side: s,
chain_p: r4(chainP),
counter_p: counter === null ? null : r4(counter),
// Recorded, not recomputed downstream: the divergence is the first signal of
// whether the chain is finding anything the counter is not.
divergence: counter === null ? null : r4(chainP - counter),
};
}
module.exports = {
runShadow, buildAtoms, alignToSide, shadowKey, upChainFn,
SHADOW_VERSION, SHADOW_STAT,
};
+19 -3
View File
@@ -76,8 +76,12 @@ async function build(deps = {}) {
let games = [];
try {
[lineups, games] = await Promise.all([
// `batting_order` rides along on a read that already happens — it is the
// OPPORTUNITY term the chain needs (PA per game is set almost entirely by
// lineup slot), and fetching it separately would be a second query for a
// column already in the row. Nothing on the served path reads it.
paginate(() => sb.from('lineup_context')
.select('as_of_date, game_date, sport, game_pk, team, side, player_key')
.select('as_of_date, game_date, sport, game_pk, team, side, player_key, batting_order')
.eq('sport', 'mlb').eq('game_date', gameDate).lte('as_of_date', asOf),
{ key: uniqueKeyFor('lineup_context'), pageSize: 1000, label: 'matchupKeys:lineup_context' }),
getSchedule(gameDate),
@@ -122,15 +126,27 @@ async function build(deps = {}) {
*/
function resolve(prop) {
const key = nameKey(prop && (prop.player || prop.player_name));
const empty = { opponent: null, opposing_pitcher: null, source: null, refused: 'no_lineup_row' };
const empty = { opponent: null, opposing_pitcher: null, batting_order: null, source: null, refused: 'no_lineup_row' };
if (!key) return { ...empty, refused: 'no_player' };
const lu = teamByPlayer.get(key);
if (!lu || !lu.team) return empty;
// The posted batting slot, carried whether or not the matchup resolves — a
// hitter with no probable pitcher declared still has a lineup position, and
// that is the opportunity term. Absent stays absent (never slot 4 by
// default, which would assert he bats cleanup).
const slot = lu.batting_order == null ? null : Number(lu.batting_order);
const battingOrder = Number.isFinite(slot) ? slot : null;
const m = matchup.get(lu.team) || matchup.get(String(lu.team).split(' ').pop());
if (!m) return { opponent: null, opposing_pitcher: null, source: null, refused: 'team_not_in_schedule' };
if (!m) {
return {
opponent: null, opposing_pitcher: null, batting_order: battingOrder,
team: lu.team, source: null, refused: 'team_not_in_schedule',
};
}
return {
opponent: m.opponent || null,
opposing_pitcher: m.opposing_pitcher || null,
batting_order: battingOrder,
team: lu.team,
source: 'lineup_context+schedule',
refused: m.opposing_pitcher ? null : 'no_probable_pitcher',
+45
View File
@@ -156,6 +156,13 @@ function rowsFromSides(base, sides, ctx = {}) {
// the multiplier is recomputable from them and is deliberately not stored.
// Null on every non-hits row and on any row with no factor context.
factor_inputs: s.factor_inputs || null,
// The SHADOW CHAIN read. Declared here (always present, usually null) so
// every row in a batch carries the same keys — PostgREST builds a bulk
// insert from the FIRST row's shape, so a column that appears only on some
// rows is silently dropped for the whole batch. Filled by
// `mergeChainShadow` after enrichment; never read by anything served.
chain_shadow: null,
});
}
return rows;
@@ -246,6 +253,43 @@ function mergeEnrichment(rows, enrichedGrades) {
});
}
/**
* Attach the SHADOW CHAIN read to collected rows — the (chain_p, counter_p,
* outcome) triple's first two thirds.
*
* Like `mergeEnrichment` this fills ONE field and touches nothing else. It runs
* later than grade time for the same structural reason the archetype does: the
* chain needs the statcast rows and park factors the enrichment pass loaded, and
* pulling that forward into the grader would put per-prop I/O on the serving
* path.
*
* THE SIDE ALIGNMENT IS THE LOAD-BEARING PART. The chain computes P(over the
* line); `p_win` on the row is expressed for the graded SIDE. Storing the raw
* over-probability against an under row's `p_win` would invert every comparison
* made from it afterwards, silently — so `alignToSide` is given the row's own
* side and the row's own counter probability, and both halves of the triple end
* up pointing the same way. `outcome` is written by the ordinary settle pass and
* is already side-aligned, which completes it.
*
* A row the chain could not read is left NULL. It is not a zero probability and
* not an average hitter — the chain refusing to read someone is a fact worth
* keeping, and a fabricated third of a triple would poison the adjudication this
* column exists to enable.
*/
function mergeChainShadow(rows, shadow) {
if (!Array.isArray(rows) || !rows.length) return rows || [];
const byKey = shadow && shadow.byKey;
if (!byKey || typeof byKey.get !== 'function') return rows;
const cs = require('./model/chainShadow');
return rows.map((r) => {
const block = byKey.get(cs.shadowKey(r.player_key, r.stat, r.line));
if (!block) return r;
const aligned = cs.alignToSide(block, r.side, r.p_win);
return aligned ? { ...r, chain_shadow: aligned } : r;
});
}
function newSnapshotId() {
return crypto.randomUUID();
}
@@ -257,6 +301,7 @@ module.exports = {
rowsFromSides,
createCollector,
mergeEnrichment,
mergeChainShadow,
persist,
newSnapshotId,
__internals: { numOrNull, intOrNull, boolOrNull },
+168 -5
View File
@@ -326,6 +326,18 @@ const CALIBRATION_BASIS = Object.freeze({});
async function runSnapshot(sport, opts = {}) {
const sp = String(sport || '').toLowerCase();
const deps = {
// ── THE SEAMS THIS OBJECT WAS AN ALLOWLIST FOR ──────────────────────────
// Fourteen call sites below read `deps.challenger`, `deps.loadStatcast`,
// `deps.environmentContext`, `deps.lineupContext`, `deps.hitsFactorContext`,
// `deps.matchupKeys`, `deps.gameBinder` and friends, each documented as
// injectable. None of them were in this literal, so every one of them was
// permanently `undefined` and always fell through to the real module — the
// injection seam existed in the comment and not in the code.
//
// Spreading FIRST rather than adding fourteen entries: every explicit key
// below still wins (they are declared after and already read `opts.X`), so
// no existing behaviour moves, while an unlisted dep now actually arrives.
...opts,
getOdds: opts.getOdds || require('./oddsService').getOdds,
gradeAndCacheSlate: opts.gradeAndCacheSlate || require('./gradeSlateService').gradeAndCacheSlate,
resolveStats: opts.resolveStats || require('./playerIntelService').resolvePlayerStats,
@@ -474,7 +486,11 @@ async function runSnapshot(sport, opts = {}) {
if (sp === 'mlb') {
try {
const ctxSvc = deps.hitsFactorContext || require('./model/hitsFactorContext');
const sbc = require('../utils/supabase').getSupabaseServiceClient();
// Injectable so the seam is REACHABLE. Without this the whole factor +
// hand-split path is gated behind a real Supabase client, so no test can
// reach it and "it is wired" would rest on reading the code — which is
// precisely how A5 shipped three factors that never fired.
const sbc = deps.supabase || require('../utils/supabase').getSupabaseServiceClient();
factorContext = sbc ? await ctxSvc.build(sbc) : null;
if (factorContext) console.log(`[factors] ${sp} hits context loaded — ${JSON.stringify(factorContext.__stats)}`);
else console.log(`[factors] ${sp} — no factor context; grading unadjusted`);
@@ -493,7 +509,7 @@ async function runSnapshot(sport, opts = {}) {
if (sp === 'mlb') {
try {
const mk = deps.matchupKeys || require('./model/matchupKeys');
const sbm = require('../utils/supabase').getSupabaseServiceClient();
const sbm = deps.supabase || require('../utils/supabase').getSupabaseServiceClient();
const mlbAdapter = deps.mlbAdapter || require('./adapters/mlbStatsAdapter');
const resolver = sbm ? await mk.build({
sb: sbm,
@@ -531,12 +547,20 @@ async function runSnapshot(sport, opts = {}) {
// empty-slate early return: a slate that graded nothing but refused
// everything is exactly the case worth recording.
let retentionRows = 0;
const persistRetention = async (enrichedGrades) => {
const persistRetention = async (enrichedGrades, chainShadow = null) => {
if (!retention || !collector || !collector.rows.length) return;
try {
const rows = retention.mergeEnrichment
let rows = retention.mergeEnrichment
? retention.mergeEnrichment(collector.rows, enrichedGrades || [])
: collector.rows;
// CHAIN v1 SHADOW — attaches the (chain_p, counter_p) pair; `outcome`
// completes the triple at settle. Own guard: the shadow is a measurement
// layer and must never cost the retention write, which is the record.
if (chainShadow && retention.mergeChainShadow) {
try { rows = retention.mergeChainShadow(rows, chainShadow); } catch (e) {
console.warn('[chain-shadow] merge skipped (retention continues):', e.message);
}
}
const r = await retention.persist(rows);
retentionRows = r.written || 0;
console.log(`[snapshot] retention ${sp}: ${r.written}/${r.attempted} rows${r.skipped ? ' (skipped — no supabase env)' : ''}${r.error ? ` ERROR: ${r.error}` : ''}`);
@@ -708,10 +732,17 @@ async function runSnapshot(sport, opts = {}) {
// `enriched` (champion p_win) is READ, never written: the serving projection
// is untouched, and the challenger rides alongside it to the ledger.
let withChallenger = enriched;
// CHAIN v1 SHADOW reuses what the challenger pass already fetched — the
// statcast profiles and the resolved opposing starter. Captured out here so
// the shadow costs ZERO additional I/O; a measurement layer that adds a fetch
// per prop inside a cron slot is how measurement becomes an outage.
let statcastRowsByKey = null;
let oppPitcherByTeam = null;
try {
const challenger = deps.challenger || require('./challengerProjection');
const axes = deps.archetypeAxes || require('./archetypeAxes');
const rowsByKey = await (deps.loadStatcast || loadStatcastRows)(sp);
statcastRowsByKey = rowsByKey;
if (rowsByKey && rowsByKey.size) {
const classifyFor = (playerName) => {
const row = rowsByKey.get(nameKey(playerName || ''));
@@ -728,6 +759,7 @@ async function runSnapshot(sport, opts = {}) {
origin: process.env.BACKEND_SELF_ORIGIN || 'http://localhost:3000',
});
ctxInternals = ctx._internals || null;
oppPitcherByTeam = (ctxInternals && ctxInternals.oppPitcherByTeam) || null;
// Enrich each grade with the hitter hand the platoon estimate needs
// (statcast_aggregates.bats, already loaded above).
const handOf = (name) => {
@@ -868,6 +900,119 @@ async function runSnapshot(sport, opts = {}) {
}
}
// ── CHAIN v1 — THE SHADOW READ ──────────────────────────────────────────
// The chain computes a probability the way the model was always described as
// working: base-event atoms chained over opportunity, rather than a count of
// how often the player has cleared this number before. It runs on the live
// board and REACHES NOTHING. `enriched` is read, never written; the served
// `p_win` is byte-identical whether or not this block executes, which is the
// same fence A6 used for the factor shadow and the same reason it is safe to
// run an uncalibrated forecast at all.
//
// `requireCalibrated:false` is legitimate here and ONLY here: no consumer
// exists. Every stored block says so in its own payload (`servable: false`),
// because a caveat that lives only in a comment is not attached to the data
// once the data is queried by something else.
//
// It writes ONE column on the retention row — the chain's probability next to
// the counter's, so the settle pass completes an adjudicable triple.
let chainShadow = null;
if (sp === 'mlb') {
try {
const cs = deps.chainShadow || require('./model/chainShadow');
const reg = deps.featureRegistry || require('./model/featureRegistry');
// CANDIDATE features, not PROVEN. The registry's live gate returns PROVEN
// only and would refuse every row — correctly, for anything served. A
// challenger must be allowed to READ what it is being measured on, which
// is exactly what `candidateFeatures` is for.
const allowed = reg.candidateFeatures('mlb');
const pitcherRowFor = (g) => {
if (!oppPitcherByTeam || !statcastRowsByKey) return null;
try {
const { abbrOf } = require('./environmentContext');
const pid = oppPitcherByTeam.get(abbrOf(g.team));
if (pid == null) return null;
for (const row of statcastRowsByKey.values()) {
if (Number(row.source_id) === Number(pid) && row.role === 'pitcher') return row;
}
return null;
} catch { return null; }
};
// ── CHAIN v2 — THE HAND SPLIT, AS-OF-CORRECT ────────────────────────
// `paOutcome` reads a hitter's SEASON rates, which already average his
// platoon split over whichever hands he happened to face. Un-averaging it
// is what makes the read a MATCHUP rather than a season.
//
// Both halves are ALREADY LOADED, so this costs nothing new: the batter's
// hand and his vs-LHP/vs-RHP splits come from `factorContext`
// (`hitsFactorContext.build`, whose reads are A4-dated), and the opposing
// starter's hand comes through the A6 `matchupKeys` resolve. Without those
// keys the resolver's `throws` is null — the A5 finding — so the split
// would silently never fire, which is the whole failure mode this order
// exists inside. Absent ⇒ NO adjustment and a recorded reason; never a
// league-typical split asserted about a hitter we cannot read.
// Gated on `factorContext` ALONE, not on both. Without the keys the
// hitter's own split is still readable and only the pitcher hand is
// missing — so the recorded reason is `no_pitcher_hand`, which names the
// input that is actually absent, rather than `no_splits`, which would
// point at the half that was there all along. That mis-naming is how A5's
// real cause stayed hidden for months.
const handSplitFor = factorContext
? (g) => {
try {
const keys = matchupKeys ? (matchupKeys(g) || null) : null;
const ctx = factorContext(g, sp, keys);
if (!ctx) return null;
return { bats: ctx.bats || null, throws: ctx.throws || null, platoonSplits: ctx.platoonSplits || null };
} catch { return null; }
}
: null;
if (!handSplitFor) {
console.log(`[chain-shadow] ${sp} — no factor context; the chain reads SEASON rates`);
} else if (!matchupKeys) {
console.log(`[chain-shadow] ${sp} — factor context loaded but NO join keys; the hand split will refuse with no_pitcher_hand`);
}
// ── CHAIN v3 — THE OPPORTUNITY TERM, REALLY FED ─────────────────────
// v2 shipped the shadow WITHOUT a `lineupSlotFor`, so every hitter fell to
// `skillProjection.DEFAULT_PA` (4.1) — "a regular" asserted about the
// leadoff man and the nine hole alike. The chain is rate x opportunity;
// holding opportunity constant across the lineup throws away the half of
// the model that is a free, known, pre-game fact.
//
// The slot comes off the A6 lineup read (`matchupKeys`), which already
// fetches the row — one extra column, no extra query, as-of dated. Absent
// ⇒ DEFAULT_PA and the block records `default_regular`, so a season-shaped
// opportunity term is never mistaken for a posted one.
const lineupSlotFor = matchupKeys
? (g) => {
try {
const k = matchupKeys(g);
return k && k.batting_order != null ? k.batting_order : null;
} catch { return null; }
}
: null;
chainShadow = cs.runShadow(withChallenger, {
statcastByKey: statcastRowsByKey,
pitcherRowFor,
handSplitFor,
lineupSlotFor,
allowed,
});
const s = chainShadow.summary;
console.log(`[chain-shadow] ${sp} — ${s.readable}/${s.atoms} atoms read (${s.refused} refused) across ${s.games_read}/${s.games} games; `
+ `hand split fired on ${s.platoon_applied}, posted lineup slot on ${s.opportunity_posted}/${s.props}; `
+ 'UN-SERVABLE, served p_win untouched');
if (s.refused) console.log(`[chain-shadow] ${sp} refusals: ${JSON.stringify(s.refusal_reasons)}`);
if (s.readable > s.platoon_applied) console.log(`[chain-shadow] ${sp} season-rate rows: ${JSON.stringify(s.platoon_reasons)}`);
} catch (e) {
// A measurement layer must never break the pipeline it measures inside.
console.warn(`[chain-shadow] ${sp} skipped:`, e.message);
chainShadow = null;
}
}
// LINEUP + BASERUNNER CONTEXT — the input RBI and runs have always needed.
// Best-effort and dated: a context failure must never break a snapshot, and a
// lineup is a PRE-GAME fact that changes by the hour, so what we knew at grade
@@ -885,7 +1030,7 @@ async function runSnapshot(sport, opts = {}) {
}
}
await persistRetention(enriched);
await persistRetention(enriched, chainShadow);
// Line deltas vs the previous snapshot's locked lines.
const prev = await deps.cacheGet(`snapshot:${sp}:latest`);
@@ -946,6 +1091,24 @@ async function runSnapshot(sport, opts = {}) {
// writes nothing. Retention is best-effort by design, which makes a broken
// write invisible without this.
retentionRows,
// Chain v1 shadow — counts only, so a run that read nothing is visible
// instead of looking like a slate with no hits props. Never a probability:
// this is the return value of a snapshot, which surfaces are allowed to read.
chainShadow: chainShadow ? {
atoms: chainShadow.summary.atoms,
readable: chainShadow.summary.readable,
refused: chainShadow.summary.refused,
props: chainShadow.summary.props,
// Reported SEPARATELY from `readable` on purpose: "the chain produced a
// number" and "the chain read tonight's matchup" are different claims, and
// collapsing them is how a season read gets reported as a forward one.
platoon_applied: chainShadow.summary.platoon_applied,
// rate x opportunity — BOTH halves reported. v2 shipped with the
// opportunity half a constant, and nothing in the summary could show it.
opportunity_posted: chainShadow.summary.opportunity_posted,
games_read: chainShadow.summary.games_read,
servable: false,
} : null,
topGrades: enriched.filter((g) => isTopGrade(g.grade)).slice(0, 5).map((g) => ({
player: g.player || g.player_name, stat: g.stat_type || g.stat, grade: g.grade, archetype: g.archetype,
})),
+215
View File
@@ -0,0 +1,215 @@
'use strict';
/**
* wnbaUsageService — ingest the WNBA possession/usage feed, and read it AS-OF.
*
* Spec: specs/wnba-possession-feed.md
*
* Two responsibilities, deliberately in one place because they must agree about
* what "as of" means:
*
* ingestRange() pull completed games and persist per-player rows
* profileAsOf() the point-in-time read the chainFn will consume
*
* ── THE AS-OF RULE, STATED ONCE ──────────────────────────────────────────
* A profile as of date D is built from games with `game_date < D`, STRICTLY.
* Not `<=`. A game on D may have tipped after grade time, so including it would
* feed tonight's result into tonight's forecast — the same leak
* `snapshotSettlementService.isPreGame` exists to stop, and the reason its drain
* had to filter by pre-game rather than by calendar date.
*
* This is the whole point-in-time story. There is no history table because
* nothing is overwritten: a completed box score is immutable, so the filter IS
* the retention.
*
* ── WHAT IT REFUSES ──────────────────────────────────────────────────────
* No games before the cutoff ⇒ `null`, not a league-average player. A replacement
* profile would silently assert a usage rate about someone we have never seen,
* and usage feeds the chain's opportunity term directly — the same reason
* `platoonSeverity` refuses a thin split rather than shrinking it to league.
*/
const adapter = require('./adapters/espnWnbaAdapter');
const { paginate } = require('../utils/safePaginate');
const { uniqueKeyFor } = require('../utils/tableKeys');
const { nameKey } = require('../utils/playerName');
const TABLE = 'wnba_player_game';
/** Minimum games before a profile is worth calling a profile. */
const MIN_GAMES = 3;
const mean = (xs) => (xs.length ? xs.reduce((a, b) => a + b, 0) / xs.length : null);
const round = (v, d = 3) => (v == null || !Number.isFinite(v) ? null : Math.round(v * 10 ** d) / 10 ** d);
/** ET calendar date, the only calendar this pipeline uses. */
function dateET(d = new Date()) {
return new Intl.DateTimeFormat('en-CA', {
timeZone: 'America/New_York', year: 'numeric', month: '2-digit', day: '2-digit',
}).format(d instanceof Date ? d : new Date(d));
}
function* eachDate(from, to) {
const start = new Date(`${from}T12:00:00Z`);
const end = new Date(`${to}T12:00:00Z`);
for (let d = start; d <= end; d = new Date(d.getTime() + 86_400_000)) {
yield d.toISOString().slice(0, 10);
}
}
/**
* Ingest every completed game in an ET date range.
*
* Idempotent: the primary key is (game_id, source_id) and this upserts, so a
* re-run over the same dates rewrites identical rows rather than duplicating.
* That matters because the natural operation here is "re-run yesterday", and a
* feed that doubles on a retry is a feed nobody dares re-run.
*/
async function ingestRange(sb, { from, to, fetchImpl, onProgress } = {}) {
const out = {
dates: 0, games: 0, rows: 0, written: 0, skipped: [], errors: [],
};
if (!sb) return { ...out, errors: ['no supabase client'] };
if (!from || !to) return { ...out, errors: ['from and to required'] };
for (const date of eachDate(from, to)) {
out.dates += 1;
let ids = [];
try {
ids = await adapter.getFinalGameIds(date, { fetchImpl });
} catch (e) {
// A failed DATE is surfaced, never swallowed as "no games" — an empty feed
// and a broken feed look identical downstream, and that costume has cost
// this codebase two outages (the fielding_oaa 404, the settle 500-row URL).
out.errors.push(`${date}: ${e.message}`);
continue;
}
for (const gameId of ids) {
let parsed;
try {
parsed = await adapter.getGameRows(gameId, { fetchImpl });
} catch (e) {
out.errors.push(`${date}/${gameId}: ${e.message}`);
continue;
}
if (parsed.skipped) { out.skipped.push({ game_id: gameId, reason: parsed.skipped }); continue; }
if (!parsed.rows.length) { out.skipped.push({ game_id: gameId, reason: 'no_player_rows' }); continue; }
out.games += 1;
const rows = parsed.rows
.filter((r) => r.source_id && r.player_name)
.map((r) => ({ ...r, player_key: nameKey(r.player_name) }));
out.rows += rows.length;
const { error } = await sb.from(TABLE).upsert(rows, { onConflict: 'game_id,source_id' });
if (error) out.errors.push(`${gameId}: ${error.message}`);
else out.written += rows.length;
if (typeof onProgress === 'function') onProgress({ date, gameId, rows: rows.length });
}
}
return out;
}
/**
* Every stored game for one player STRICTLY BEFORE `asOf`.
*
* Read through `safePaginate`: an unordered `.range()` returns the right COUNT
* and the wrong ROWS, measured at up to 33.6% duplication on this database, and
* duplicated games would double-weight a player's usage.
*/
async function gamesBefore(sb, { playerKey, asOf, season = null } = {}) {
if (!sb || !playerKey || !asOf) return [];
return paginate(
() => {
let q = sb.from(TABLE).select('*').eq('sport', 'wnba')
.eq('player_key', playerKey).lt('game_date', asOf);
if (season != null) q = q.eq('season', season);
return q;
},
{ key: uniqueKeyFor(TABLE), label: 'wnbaUsage:gamesBefore' },
);
}
/**
* THE POINT-IN-TIME PROFILE the chainFn will consume.
*
* @returns {object|null} null when fewer than `minGames` games precede `asOf` —
* never a league-average stand-in.
*/
async function profileAsOf(sb, { playerKey, asOf, season = null, minGames = MIN_GAMES, recent = 5 } = {}) {
const games = await gamesBefore(sb, { playerKey, asOf, season });
if (games.length < minGames) {
return null;
}
// Most recent first — the recency window is the head of this list.
games.sort((a, b) => String(b.game_date).localeCompare(String(a.game_date)));
const played = games.filter((g) => Number(g.minutes) > 0);
const usable = played.filter((g) => g.usage_rate != null);
const recentGames = played.slice(0, Math.max(1, recent));
// MINUTES-WEIGHTED usage, not a flat mean. A 6-minute cameo and a 34-minute
// start are one row each; weighting by playing time is the same lesson the
// lineup K-rate learned — an unweighted team aggregate counts a 12-PA callup
// like an everyday starter, and it HURT the model until it was PA-weighted.
const wUsage = (() => {
const w = usable.reduce((a, g) => a + Number(g.minutes), 0);
if (!(w > 0)) return null;
return usable.reduce((a, g) => a + Number(g.usage_rate) * Number(g.minutes), 0) / w;
})();
return {
player_key: playerKey,
as_of: asOf,
// The cutoff is reported so a caller can see the read was bounded, and by what.
games: games.length,
games_played: played.length,
last_game_date: games[0] ? games[0].game_date : null,
// ── the chainFn's three inputs ──
usage_rate: round(wUsage, 3),
usage_rate_recent: round(mean(recentGames.filter((g) => g.usage_rate != null).map((g) => Number(g.usage_rate))), 3),
minutes_per_game: round(mean(played.map((g) => Number(g.minutes))), 2),
minutes_recent: round(mean(recentGames.map((g) => Number(g.minutes))), 2),
team_pace: round(mean(played.filter((g) => g.team_pace != null).map((g) => Number(g.team_pace))), 3),
team_possessions: round(mean(played.filter((g) => g.team_possessions != null).map((g) => Number(g.team_possessions))), 3),
ts_pct: round(mean(played.filter((g) => g.ts_pct != null).map((g) => Number(g.ts_pct))), 4),
efg_pct: round(mean(played.filter((g) => g.efg_pct != null).map((g) => Number(g.efg_pct))), 4),
// ── game-state, for the redistribute hook ──
starter_rate: round(played.length ? played.filter((g) => g.starter).length / played.length : null, 3),
avg_final_margin: round(mean(played.filter((g) => g.final_margin != null).map((g) => Number(g.final_margin))), 2),
team: games[0] ? games[0].team : null,
source: 'espn_wnba_boxscore',
};
}
/** Coverage of what is stored — rows, players, games, date range. */
async function coverage(sb, { season = null } = {}) {
if (!sb) return null;
const rows = await paginate(
() => {
let q = sb.from(TABLE).select('game_id,source_id,player_key,game_date,season,usage_rate,team_pace,minutes')
.eq('sport', 'wnba');
if (season != null) q = q.eq('season', season);
return q;
},
{ key: uniqueKeyFor(TABLE), label: 'wnbaUsage:coverage' },
);
if (!rows.length) return { rows: 0 };
const dates = rows.map((r) => r.game_date).filter(Boolean).sort();
return {
rows: rows.length,
players: new Set(rows.map((r) => r.player_key)).size,
games: new Set(rows.map((r) => r.game_id)).size,
from: dates[0],
to: dates[dates.length - 1],
with_usage: rows.filter((r) => r.usage_rate != null).length,
with_pace: rows.filter((r) => r.team_pace != null).length,
};
}
module.exports = {
ingestRange, profileAsOf, gamesBefore, coverage, dateET, eachDate,
TABLE, MIN_GAMES,
};
+6
View File
@@ -38,6 +38,12 @@ const UNIQUE_KEY = Object.freeze({
park_dimensions: Object.freeze(['as_of_date', 'sport', 'venue_id']),
hitter_opportunity: Object.freeze(['as_of_date', 'sport', 'season', 'player_key']),
lineup_context: Object.freeze(['as_of_date', 'sport', 'game_pk', 'player_key']),
// PER-GAME grain, not a dated snapshot. A completed box score is immutable, so
// point-in-time here is `game_date < asOf` over facts that never change —
// there is no aggregate being overwritten and therefore no history table to
// pair it with. (migration 039)
wnba_player_game: Object.freeze(['game_id', 'source_id']),
});
/**