diff --git a/BLOCKERS.md b/BLOCKERS.md index b32b46f..03e553e 100755 --- a/BLOCKERS.md +++ b/BLOCKERS.md @@ -24,3 +24,22 @@ **Workaround:** Apply migration via Supabase Dashboard SQL Editor **Resolution path:** Either DNS propagates, or add Supabase IP to /etc/hosts, or use `supabase link` with access token via api.supabase.com **Owner:** Kev + + +## PINNACLE MLB PROP COVERAGE STOPPED — 2026-07-31 (open, external) + +**One question to PropLine:** *why did Pinnacle MLB player-prop coverage stop on +2026-07-31?* + +**Evidence:** `closing_captures` MLB, pinnacle: 103,940 rows over 07-20 → 07-30, +then **4,022 → 0** on 07-31 and zero since, while every other book continued +normally (07-31: 47,606 non-pinnacle rows). `line_type='sharp'` is a label applied +in `closingCapture.js` via `SHARP_BOOKS` — same feed, not a separate provider. + +**Why it matters:** Pinnacle is the only sharp anchor we have ever had for props +(17,090 two-sided captures in that window). Without it the consensus ruler is a +**market** consensus, not a **sharp** one. + +**Status:** `market-not-sharp` is **PENDING-RECOVERY, not confirmed permanent.** +Do NOT enshrine it in MASTER-PLAN as a permanent limitation until this is +answered. Not caused by any VYNDR change — the display widening only adds books. diff --git a/specs/MASTER-PLAN.md b/specs/MASTER-PLAN.md index 0901406..9efadcc 100644 --- a/specs/MASTER-PLAN.md +++ b/specs/MASTER-PLAN.md @@ -46,8 +46,18 @@ Two things gate on accrual instead of on code, and cannot be rushed: - The **edge verdict** may change under that ruler — or may not; today it is unproven either way, and 'unproven' is the honest label. -**Open, cheap, and unrelated to the model:** the 🔴 pinnacle feed regression, and -the display layer still shows nothing of the widened multi-book data. +**DONE 2026-08-01** — the challenger is built and measured +(`specs/rank-on-pwin-challenger.md`): `rankByForecast` (p_win-first, **no edge +term**), delta recorded (MLB **87.5% of props move, top read changes**), edge +retired from decisions and from display quality-signalling, `forecast_rank` +stamped additively, **MLB-only by an enforced `FORECAST_RANKED_SPORTS` guard**. + +**THE FLIP IS THE NEXT ORDER** — one decision, three edits. **Re-run the delta on +a full MLB slate first**; it measured on 8 graded props. + +**Open, cheap, unrelated to the model:** the 🔴 pinnacle question (logged in +`BLOCKERS.md`), and the display layer still shows nothing of the widened +multi-book data. --- @@ -58,7 +68,7 @@ the display layer still shows nothing of the widened multi-book data. | **MLB slate invisible to us** | **64.8%** | our own allow-list, not the feed — now widened for DISPLAY | | **books/prop, MLB** | 3.61 feed → 0.57 after filter | the filter cost, quantified | | **books/prop, WNBA** | **4.21** feed → 1.20 | **WNBA is BETTER covered than MLB** | -| **consensus ruler** | **MARKET, not SHARP** | `pinnacle`/`matchbook`/`polymarket` = **0%** on both sports. No sharp anchor exists in our feed. Permanent limitation, not a milestone | +| **consensus ruler** | **MARKET, not SHARP — ⏳ PENDING-RECOVERY, not permanent** | `matchbook`/`polymarket` = 0%, but **`pinnacle` ran until 07-30** (103,940 captures) and stopped. **Do not enshrine as permanent** until PropLine answers — see `BLOCKERS.md` | | **ruler delta** (consensus − incumbent) | MLB mean +1.50 pts, median 0, **17% of props move ≥5 pts** | rulers genuinely differ; "better" is unproven | | **MLB isotonic `p_win`** | **DECIDED** — reliability **0.0846**, resolution **0.190**, holdout **n=125** | **PROVISIONAL label RETRACTED 2026-08-01.** Calibration is **ruler-independent** (`estimateProbability` never sees a price; the fit is p_win-vs-outcome). Replicated on a fresh later window, both metrics improved | | **edge vs the ruler** | corr(edge, outcome) **−0.010** (v1) → **−0.022** (v2), n=200 · corr(**p_win**, outcome) **+0.26** | **Subtracting the market DESTROYS the signal.** The consensus ruler does not rescue edge: *differs ≠ better* | diff --git a/specs/rank-on-pwin-challenger.md b/specs/rank-on-pwin-challenger.md new file mode 100644 index 0000000..1ffc472 --- /dev/null +++ b/specs/rank-on-pwin-challenger.md @@ -0,0 +1,153 @@ +# RANK ON p_win — CHALLENGER DELTA + EDGE RETIREMENT + +**Date:** 2026-08-01 · **Live ordering byte-identical** · challenger measured on +live prod grades · edge retired from decisions and from display quality-signalling. + +**Gates:** 4,039 tests / 323 suites green · `next build` exit 0 · delta recorded · +live sorts untouched. + +--- + +## WHY (the measurement that dictates this) + +| instrument | corr with outcome, n=200 settled MLB | +|---|---:| +| **`p_win`** | **+0.26** | +| `p_win − fair_prob` (v1 single-book ruler) | −0.010 | +| `p_win − fair_prob` (v2 consensus ruler) | −0.022 | + +**Subtracting the market destroys the signal, under both rulers.** A quantity +that does not predict must not rank, gate, or decide. + +--- + +## THE CHALLENGER — `rankByForecast` + +**Order:** takeable-gated `p_win` → grade tier → confidence → stable input order. +**No edge term anywhere** (a test flips edge from −99 to +99 and asserts the +order does not move). + +**`p_win` leads and the letter follows — deliberately.** The grade letter measured +**r ≈ 0.005** against outcomes and is **inverted** (B 52.4% < C 56.9%), while +`p_win` measures **+0.26**. Leading with the letter would sort the board by the +weaker signal and use the stronger one only to break ties. + +The takeable gate is unchanged and mandatory: raw `p_win` crowns −300 chalk. + +### One thing worth stating precisely + +**Isotonic calibration is a MONOTONE transform, so ranking on raw `p_win` and on +calibrated `p_win` produce the SAME ORDER.** Calibration matters when `p_win` is +*displayed* or *thresholded* — it cannot change a ranking. The order asked to +"rank on calibrated p_win"; for ranking specifically, that is a no-op relative to +raw. Recorded in the code so nobody re-derives it. + +--- + +## THE DELTA (live prod grades, nothing flipped) + +`GET /api/internal/ranking-delta` · `live_ordering_unchanged: true` + +| | MLB | WNBA | +|---|---:|---:| +| graded props | 8 | 25 | +| `p_win` coverage | 100% | 100% | +| **props that move** | **7/8 (87.5%)** | **25/25 (100%)** | +| mean \|move\| | 2.5 places | 4.1 places | +| max move | 5 | 12 | +| top-10 overlap | 80% | 80% | +| **top read changes** | **YES** | **YES** | + +MLB #1: `corey seager | hits 1.5 under` → `jake burger | hits 0.5 over` +(Seager falls 1 → 6). + +**This is a large re-ordering, not a tweak.** Nearly every row moves and the +headline read changes. + +**Caveat, stated because it matters:** MLB's slate had only **8 graded props** at +measurement time. The percentages are real but the sample is a single small +slate — re-run the endpoint on a full slate before the flip. It is one call. + +--- + +## PER-SPORT DOCTRINE — ENFORCED IN CODE, NOT IN A COMMENT + +**WNBA moves the most (100% of rows, mean 4.1 places) and must NOT adopt this.** +WNBA's `p_win` is **anti-predictive** on its own data — it abstains. Ranking that +board by `p_win` would sort it by a signal measured to point the *wrong way*: +worse than the incumbent, not better. + +A comment would not have stopped a future flip from applying this globally, so: + +```js +const FORECAST_RANKED_SPORTS = Object.freeze(new Set(['mlb'])); +``` + +`ranksOnForecast(sport)` gates the `forecast_rank` stamp, and tests assert **no +sport inherits MLB's result** — a sport joins only by passing its **own** holdout +(honest calibration AND surviving resolution). + +**The WNBA number above is informational only.** It is in the report to show what +the guard is preventing. + +--- + +## WHAT CHANGED, WHAT DID NOT + +### Changed now (not rankings, so not gated on the flip) + +- **`altLineScanner.compareToBookImplied`** no longer returns + `value_detected: edge > 0`. **Edge is still computed and returned** — losing the + record would be worse than mis-using it — but the verdict is `null` with + `value_basis: 'retired:edge_does_not_predict'`. +- **`scanAltLines`** no longer filters to `edge > 0` nor calls the survivor + `optimal_line`. The whole ladder returns, ranked, labelled + `price_gap_diagnostic_unvalidated`. **The module has zero callers** (verified) — + unwired like `mlbGrader.js`; left in place and made honest rather than deleted. +- **`MobileEdgeBoard.EdgeCell`** no longer renders green for positive edge and red + for negative. Two things were wrong with that: green/red **is** a quality claim + on a quantity that does not predict, and **ROW-GRAMMAR reserves red for + settled-negative only** (miss/dead/stale/faded) — a negative diagnostic is not a + settled loss. Now neutral mono with a diagnostic tooltip; the column header + reads **`MKT GAP · DIAGNOSTIC`**. **The number is still shown** — no display went + blank. +- **`DeskShowcase`** edge colour neutralised for the same reason. + +### An honest asymmetry I did not paper over + +**Ranking props against each other must not use edge. Choosing between RUNGS of +the same prop is inherently price-relative** — ranking rungs by model probability +alone would always pick the lowest line, because P(over 0.5) > P(over 2.5) by +construction. So the gap stays the rung key in `scanAltLines`, **explicitly +labelled unvalidated**, rather than being replaced by something that would look +principled and be degenerate. + +### NOT changed (challenger-first) + +- **`rankGrades`** — the incumbent (grade-first, edge as 4th key) is untouched, + and tested as untouched. +- **Every live sort** still calls the incumbent. `selectTopGrades`, + `flattenToEdgeBoard`, `topGradedService` — all byte-identical. +- **`forecast_rank`** is stamped additively on MLB snapshot grades, **before** + `stripModelPrice`, so every tier would receive the correct order without the + paid values (the `topGradedService` precedent — an ordinal travels where the + magnitude cannot). **Nothing sorts by it yet.** +- **Edge stays stored** in the ledger, as required. + +--- + +## THE FLIP, WHEN YOU WANT IT + +One decision, three edits: `selectTopGrades` and `flattenToEdgeBoard` sort by +`forecast_rank` when present; `rankGrades` drops its edge key. **Re-run the delta +on a full MLB slate first** — 8 props is not a slate. + +## PINNACLE — LOGGED, NOT ENSHRINED + +Per the order: **"market-not-sharp" is PENDING-RECOVERY, not a confirmed permanent +limitation.** The question for PropLine is logged in `BLOCKERS.md`: + +> *Why did Pinnacle MLB player-prop coverage stop on 2026-07-31?* Captures ran +> 103,940 over the prior 10 days, then 4,022 → 0 while every other book continued. + +Until answered, the plan must not record "no sharp anchor exists" as permanent. diff --git a/src/routes/snapshot.js b/src/routes/snapshot.js index 9ed6528..c7754df 100644 --- a/src/routes/snapshot.js +++ b/src/routes/snapshot.js @@ -20,7 +20,7 @@ const { indexRosterLogs, attachLast10Dots } = require('../services/last10Dots'); // viewer. This endpoint is PUBLIC, so the Session-66 gate on /api/analyze was // being bypassed here on every graded row. Same layer as the CLV gate. const { stripModelPrice, gateItemizedGrades, liveLockedSummary, freeSample, entitledToItemizedGrades } = require('../utils/snapshotGating'); -const { rankByForecast, gradeKey } = require('../utils/gradeRanking'); +const { rankByForecast, gradeKey, ranksOnForecast } = require('../utils/gradeRanking'); const { resolveTierFromRequest } = require('../utils/requestTier'); const router = express.Router(); @@ -121,6 +121,10 @@ router.get('/:sport', async (req, res) => { // and accepted for the top-graded board. const stampForecastRank = (grades) => { if (!Array.isArray(grades) || grades.length === 0) return grades; + // PER-SPORT DOCTRINE, enforced: only sports that passed their OWN holdout + // may be ranked by forecast. WNBA's p_win is anti-predictive, so stamping + // it would hand a future flip a wrong-way ordering for that board. + if (!ranksOnForecast(sport)) return grades; const ranked = rankByForecast(grades); const pos = new Map(); ranked.forEach((g, i) => pos.set(gradeKey(g), i + 1)); diff --git a/src/utils/gradeRanking.js b/src/utils/gradeRanking.js index 7e29a3f..bc7015d 100644 --- a/src/utils/gradeRanking.js +++ b/src/utils/gradeRanking.js @@ -141,6 +141,20 @@ function rankByForecast(grades, limit) { return limit == null ? out : out.slice(0, Math.max(0, limit)); } +/** + * WHICH SPORTS MAY RANK ON THE FORECAST — per-sport doctrine, enforced in code. + * + * MLB only. WNBA's p_win is ANTI-PREDICTIVE on its own data (it abstains), so + * ranking WNBA by p_win would sort that board by a signal measured to point the + * wrong way — worse than the incumbent, not better. A comment would not have + * stopped a future flip from applying this globally; this does. + * + * A sport joins this set only by passing its OWN holdout: honest calibration + * AND surviving resolution. Never by inheriting MLB's result. + */ +const FORECAST_RANKED_SPORTS = Object.freeze(new Set(['mlb'])); +const ranksOnForecast = (sport) => FORECAST_RANKED_SPORTS.has(String(sport || '').toLowerCase()); + /** Stable identity for a grade row, for comparing two orderings. */ function gradeKey(g) { if (!g) return ''; @@ -204,4 +218,5 @@ function rankingDelta(grades, topN = 10) { module.exports = { GRADE_RANK, gradeRankOf, strictNum, takeablePWin, descNullsLast, rankGrades, rankByForecast, rankingDelta, gradeKey, + FORECAST_RANKED_SPORTS, ranksOnForecast, }; diff --git a/tests/unit/rankingInstrument.test.js b/tests/unit/rankingInstrument.test.js index 8f1af56..fc46c91 100644 --- a/tests/unit/rankingInstrument.test.js +++ b/tests/unit/rankingInstrument.test.js @@ -10,7 +10,7 @@ */ const { - rankGrades, rankByForecast, rankingDelta, gradeKey, takeablePWin, + rankGrades, rankByForecast, rankingDelta, gradeKey, takeablePWin, ranksOnForecast, FORECAST_RANKED_SPORTS, } = require('../../src/utils/gradeRanking'); const g = (player, grade, p_win, odds = -110, extra = {}) => ({ @@ -96,3 +96,18 @@ describe('rankingDelta — the challenger-first measurement', () => { expect(JSON.stringify(rows)).toBe(snapshot); }); }); + +describe('per-sport doctrine — who may rank on the forecast', () => { + it('MLB may; WNBA may NOT (its p_win is anti-predictive, it abstains)', () => { + expect(ranksOnForecast('mlb')).toBe(true); + expect(ranksOnForecast('MLB')).toBe(true); + expect(ranksOnForecast('wnba')).toBe(false); + }); + + it('no sport inherits MLB’s result — unknown sports are excluded', () => { + for (const s of ['nba', 'nfl', 'soccer', 'nhl', 'ncaab', '', null, undefined]) { + expect(ranksOnForecast(s)).toBe(false); + } + expect([...FORECAST_RANKED_SPORTS]).toEqual(['mlb']); + }); +}); diff --git a/web/src/app/pricing/DeskShowcase.tsx b/web/src/app/pricing/DeskShowcase.tsx index 68be74f..f378924 100644 --- a/web/src/app/pricing/DeskShowcase.tsx +++ b/web/src/app/pricing/DeskShowcase.tsx @@ -31,7 +31,9 @@ const statLabel = (s?: string | null) => (s ? (STAT_LABEL[s] || s.replace(/_/g, function RungCell({ line, grade, edge, base }: Rung) { const gradeCol = grade.startsWith('A') ? 'var(--g-a)' : grade.startsWith('B') ? 'var(--text-0)' : grade.startsWith('C') ? 'var(--text-1)' : 'var(--miss)'; - const edgeCol = edge == null ? 'var(--text-2)' : edge >= 0 ? 'var(--g-a)' : 'var(--miss)'; + // Edge is a DIAGNOSTIC, not a quality signal — it does not predict outcomes + // (n=200 settled MLB: corr -0.010 / -0.022). Neutral mono, no green/red claim. + const edgeCol = 'var(--text-1)'; return (
{line}{base ? ' •' : ''}
diff --git a/web/src/components/vyndr/MobileEdgeBoard.tsx b/web/src/components/vyndr/MobileEdgeBoard.tsx index f9ad872..499ddc0 100644 --- a/web/src/components/vyndr/MobileEdgeBoard.tsx +++ b/web/src/components/vyndr/MobileEdgeBoard.tsx @@ -43,14 +43,28 @@ function rampOpacity(rank: number): number { // the pipeline's miscalibrated placeholder (seen live: 60/100/140). Rendering // "+140%" would fabricate a signal, so we show it absent. Backend fix pending. const SANE_EDGE_MAX = 40; +// EDGE IS A DIAGNOSTIC, NOT A QUALITY SIGNAL (2026-08-01). +// +// This used to render green (--g-a, the A-grade colour) for a positive edge and +// red (--miss) for a negative one. Both were wrong: +// 1. Green/red IS a quality claim, and edge does not predict outcomes — +// measured on n=200 settled MLB rows, corr(edge, outcome) = -0.010 under +// the incumbent ruler and -0.022 under the consensus ruler. +// 2. ROW-GRAMMAR reserves red for settled-negative ONLY (miss/dead/stale/ +// faded). A negative diagnostic is not a settled loss. +// It now renders in neutral mono. The number is still shown — it is a real +// measurement and worth keeping visible — it just no longer claims anything. function EdgeCell({ edge }: { edge: number | null }) { if (edge == null || Math.abs(edge) > SANE_EDGE_MAX) { return ; } - const pos = edge >= 0; return ( - - {pos ? '+' : ''}{edge.toFixed(1)}% + + {edge >= 0 ? '+' : ''}{edge.toFixed(1)}% ); } @@ -77,7 +91,7 @@ export default function MobileEdgeBoard({ {/* EDGE BOARD header */}
EDGE BOARD - EDGE ▼ + MKT GAP · DIAGNOSTIC
{/* Ranked rows */}