a8de676756
The band gate was blocked for its `else` branch. It read:
candidate = F(raw)
served = inCertifiedBand(candidate) ? candidate : RAW
and above raw 0.60 the model is measured overconfident — holdout raw 0.80-0.90
predicts 0.843 and realizes 0.639. So "the calibrator is not supported here" was
being answered with a number already proven wrong. Unsupported calibration does
not make raw true.
Four candidates were adjudicated on ONE split — fit on the earliest 60% of
train, decide support on the last 40%, evaluate on a holdout that saw neither:
A low-param 80.2% coverage 0.24374 REFUTED — its extra region
(raw 0.80-0.90) certified on cert (err +0.040, n=55) and
refuted on holdout (served 0.754 vs observed 0.639), and it
leaves a hole at 0.70-0.80 while serving the island above it
B isotonic 91.3% coverage 0.24337 CERTIFIED, contiguous raw [0.50,0.80)
C empirical band 91.3% coverage 0.24335 REFUTED — refitted point-in-time on
current-model hits the realized rates INVERT in grade order
(B+ 0.593 < B 0.614 < C+ 0.623), so the served function steps
down at raw 0.78. Its shipped constants come from 3,417 props
pooled across four batter stats and do not reproduce here
D raw identity 43.1% coverage 0.24866 certifies raw 0.50-0.60 and only there
Raw is candidate D, not a fallback. It earns exactly one region (holdout error
+0.010 on n=1,316), which is why the law is "raw must earn its region" rather
than "raw is never true". B already covers that region, so no hybrid is built.
Above raw 0.80 nothing is certified and nothing is served. That is the region
where raw is most wrong, isotonic over-corrects (cert err -0.093) and its LODO
mapping at 0.95 has spread 0.180. 8.7% of holdout rows land there.
The registry did not need changing. `serves(stat, p)` already tested certified
bands against the RAW p_win — support in the input domain, the correct question —
and returned {serve:false, reason}. It never said "serve raw". The output-space
gate and the raw fallback were both invented downstream in calibrationService.
ACTIVATION IS OFF. PROBABILITY_CONTRACT_SHADOW defaults to 0, CALIBRATION_DEPLOYED
stays frozen empty, and every served field is byte-identical. This releases the
support first, which is the required order. The shadow records raw belief, the
candidate served value, the state, the estimator identity, and what EV/Kelly/VALUE
would be under the actionability law — into its own column, read by nothing.
Migration 051 was applied to production BEFORE retentionService named the column.
PostgREST builds a bulk insert from the first row's shape, so a key whose column
does not exist 400s the whole batch silently — that is how migration 038 took
retention down for three days.
The user-facing contradiction is NOT fixed here. A B+ still says "realized about
66%" beside a confidence of 84. Fixing that is activation, and activation costs
32% of VALUE flags and 46% of Kelly recommendations on the holdout.
Suite 401/401, 5,580 passed, 4 skipped, deterministic across three runs.
Teeth 23/23, each independently injected and restored byte-identically.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CQJeAG8vcDoL5zkiaJyVb8
222 lines
9.3 KiB
JavaScript
222 lines
9.3 KiB
JavaScript
'use strict';
|
|
|
|
/**
|
|
* THE CONTRACT'S ONE JOB: never answer "the calibrator is not supported here"
|
|
* with a number we have already measured to be wrong.
|
|
*/
|
|
const pc = require('../../src/services/model/probabilityContract');
|
|
const svc = require('../../src/services/model/probabilityContractService');
|
|
const reg = require('../../src/services/model/calibrationRegistry');
|
|
|
|
const ERA = 'engine1@2026-08-07-fullwindow';
|
|
const read = (p, over = {}) => ({ sport: 'mlb', stat: 'hits', model_version: ERA, p_win: p, ...over });
|
|
// A stand-in shaped like the real map: monotone, flattening, correcting downward.
|
|
const iso = (p) => Math.round((0.42 + 0.26 * p) * 1000) / 1000;
|
|
const deps = { estimate: iso };
|
|
|
|
describe('probability object and states', () => {
|
|
it('serves a calibrated number inside certified raw support', () => {
|
|
const r = pc.resolve(read(0.65), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.CERTIFIED_CALIBRATED);
|
|
expect(r.served_probability).toBe(iso(0.65));
|
|
expect(pc.isCertified(r)).toBe(true);
|
|
});
|
|
|
|
it('NO RAW FALLBACK — an uncertified region serves no number at all', () => {
|
|
for (const p of [0.85, 0.90, 0.95, 0.99]) {
|
|
const r = pc.resolve(read(p), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.UNCERTIFIED);
|
|
expect(r.served_probability).toBeNull();
|
|
// the specific defect: served must not silently become the raw value
|
|
expect(r.served_probability).not.toBe(p);
|
|
}
|
|
});
|
|
|
|
it('raw probability is preserved in every state, including refusals', () => {
|
|
for (const p of [0.45, 0.55, 0.85, 0.95]) {
|
|
expect(pc.resolve(read(p), deps).raw_model_probability).toBe(p);
|
|
}
|
|
expect(pc.resolve(read(0.9), deps).raw_model_probability).toBe(0.9);
|
|
});
|
|
|
|
it('a different model era is a different forecaster — VERSION_MISMATCH, no number', () => {
|
|
const r = pc.resolve(read(0.65, { model_version: 'engine1@2026-07-20' }), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.VERSION_MISMATCH);
|
|
expect(r.served_probability).toBeNull();
|
|
});
|
|
|
|
it('another stat or sport is UNSUPPORTED, never quietly served', () => {
|
|
for (const o of [{ stat: 'total_bases' }, { stat: 'rbi' }, { sport: 'wnba' }, { sport: 'nba' }]) {
|
|
const r = pc.resolve(read(0.65, o), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.UNSUPPORTED);
|
|
expect(r.served_probability).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('an absent or out-of-range raw probability is INVALID, not coerced', () => {
|
|
for (const p of [null, undefined, NaN, -0.1, 1.4, 'x']) {
|
|
const r = pc.resolve(read(p), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.INVALID);
|
|
expect(r.served_probability).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('refuses when the estimator declines inside its own support', () => {
|
|
const r = pc.resolve(read(0.65), { estimate: () => null });
|
|
expect(r.probability_state).toBe(pc.STATE.UNCERTIFIED);
|
|
expect(r.served_probability).toBeNull();
|
|
});
|
|
|
|
it('the certified band is half-open and expressed in the RAW input domain', () => {
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.50)).toBe(true);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.799)).toBe(true);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.80)).toBe(false);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.499)).toBe(false);
|
|
});
|
|
});
|
|
|
|
describe('the served function is non-decreasing across its support', () => {
|
|
it('never moves backwards as raw confidence rises', () => {
|
|
let prev = null;
|
|
for (let x = 0.30; x <= 1.0001; x += 0.001) {
|
|
const p = Math.round(x * 1000) / 1000;
|
|
const r = pc.resolve(read(p), deps);
|
|
if (r.served_probability == null) continue;
|
|
if (prev != null) expect(r.served_probability).toBeGreaterThanOrEqual(prev);
|
|
prev = r.served_probability;
|
|
}
|
|
});
|
|
|
|
it('a gap is an absence, not a step down to raw', () => {
|
|
const inside = pc.resolve(read(0.799), deps).served_probability;
|
|
const outside = pc.resolve(read(0.80), deps);
|
|
expect(inside).not.toBeNull();
|
|
expect(outside.served_probability).toBeNull();
|
|
expect(outside.served_probability).not.toBe(0.80);
|
|
});
|
|
});
|
|
|
|
describe('actionability law', () => {
|
|
const odds = -115;
|
|
it('derives EV, Kelly and VALUE from the SERVED probability', () => {
|
|
const r = pc.resolve(read(0.65), deps);
|
|
const d = pc.derivedClaims(r, odds);
|
|
expect(d.available).toBe(true);
|
|
expect(d.computed_from).toBe('served_probability');
|
|
const { evPct } = require('../../src/utils/devig');
|
|
expect(d.ev_pct).toBe(evPct(r.served_probability, odds));
|
|
expect(d.ev_pct).not.toBe(evPct(0.65, odds)); // NOT from raw
|
|
});
|
|
|
|
it('withdraws EV, Kelly and VALUE entirely when no probability is certified', () => {
|
|
for (const p of [0.85, 0.95]) {
|
|
const d = pc.derivedClaims(pc.resolve(read(p), deps), odds);
|
|
expect(d.available).toBe(false);
|
|
expect(d.ev_pct).toBeNull();
|
|
expect(d.kelly).toBeNull();
|
|
expect(d.value).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('never computes a derived claim from raw behind the scenes', () => {
|
|
const { evPct } = require('../../src/utils/devig');
|
|
const { quarterKelly } = require('../../src/utils/kelly');
|
|
const d = pc.derivedClaims(pc.resolve(read(0.91), deps), odds);
|
|
expect(d.ev_pct).not.toBe(evPct(0.91, odds));
|
|
expect(d.kelly).not.toEqual(quarterKelly(0.91, odds));
|
|
expect(d.kelly).toBeNull();
|
|
});
|
|
|
|
it('a VERSION_MISMATCH withdraws actionability too', () => {
|
|
const d = pc.derivedClaims(pc.resolve(read(0.65, { model_version: 'other' }), deps), odds);
|
|
expect(d.available).toBe(false);
|
|
});
|
|
});
|
|
|
|
describe('confidence display', () => {
|
|
it('shows an exact number only when the state is certified', () => {
|
|
const c = pc.confidenceDisplay(pc.resolve(read(0.65), deps));
|
|
expect(c.exact_probability).toBe(iso(0.65));
|
|
expect(c.calibrated).toBe(true);
|
|
});
|
|
|
|
it('shows NO exact confidence when uncertified — and never the raw value', () => {
|
|
const c = pc.confidenceDisplay(pc.resolve(read(0.91), deps));
|
|
expect(c.exact_probability).toBeNull();
|
|
expect(c.exact_pct).toBeNull();
|
|
expect(c.calibrated).toBe(false);
|
|
expect(c.label).toBe('Confidence not calibrated');
|
|
expect(JSON.stringify(c)).not.toContain('0.91');
|
|
});
|
|
});
|
|
|
|
describe('the artifact records its own adjudication', () => {
|
|
it('is pinned to the current model era and the isotonic estimator', () => {
|
|
expect(pc.MLB_HITS.model_version).toBe(ERA);
|
|
expect(pc.MLB_HITS.estimator_type).toBe(pc.ESTIMATOR.ISOTONIC);
|
|
expect(pc.MLB_HITS.estimator_version).toBeTruthy();
|
|
expect(pc.MLB_HITS.certification_version).toBeTruthy();
|
|
});
|
|
|
|
it('certifies nothing above raw 0.80 — the region where raw is most wrong', () => {
|
|
for (const p of [0.80, 0.85, 0.90, 0.95]) {
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, p)).toBe(false);
|
|
}
|
|
});
|
|
|
|
it('holds ONE contract — no other sport or stat is certified', () => {
|
|
expect(Object.keys(pc.CONTRACTS)).toEqual(['mlb:hits']);
|
|
});
|
|
|
|
it('the held-out interval it records excludes zero', () => {
|
|
expect(pc.MLB_HITS.evidence.ci95[1]).toBeLessThan(0);
|
|
});
|
|
});
|
|
|
|
describe('the registry already asked the right question', () => {
|
|
it('serves() tests certified bands against the RAW p_win, not the output', () => {
|
|
const r = reg.createRegistry();
|
|
r.deploy('hits', { lodo_pass: true, ci: [-0.006, -0.001], map: { x: [0], y: [0] },
|
|
certified_bands: [[0.50, 0.80]] });
|
|
expect(r.serves('hits', 0.65).serve).toBe(true);
|
|
expect(r.serves('hits', 0.90).serve).toBe(false);
|
|
});
|
|
|
|
it('a refusal names the reason and never proposes raw', () => {
|
|
const r = reg.createRegistry();
|
|
r.deploy('hits', { lodo_pass: true, ci: [-0.006, -0.001], map: {}, certified_bands: [[0.50, 0.80]] });
|
|
const out = r.serves('hits', 0.92);
|
|
expect(out.serve).toBe(false);
|
|
expect(JSON.stringify(out)).not.toContain('0.92');
|
|
});
|
|
});
|
|
|
|
describe('probabilityContractService', () => {
|
|
it('returns null — not raw — when there is no settled history', async () => {
|
|
const built = await svc.build({}, { calibrationService: { fromLedger: async () => null } });
|
|
expect(built).toBeNull();
|
|
});
|
|
|
|
it('takes the MAP and never the blocked calibrate() gate', async () => {
|
|
const cal = require('../../src/services/model/calibration');
|
|
const map = cal.fitIsotonic(Array.from({ length: 600 }, (_, i) => {
|
|
const p = 0.40 + (i % 55) / 100;
|
|
return { p, won: i % 3 === 0 ? 0 : 1, date: `d${i % 12}` };
|
|
}), { minTotal: 200 });
|
|
const calibrateSpy = jest.fn(() => ({ p_calibrated: 0.99, calibrated: true }));
|
|
const built = await svc.build({}, { calibrationService: {
|
|
fromLedger: async () => ({ map, fit_n: 600, fitted_through: 'd11', calibrate: calibrateSpy }) } });
|
|
expect(built).not.toBeNull();
|
|
const r = built.resolve({ model_version: ERA, p_win: 0.65 });
|
|
expect(r.probability_state).toBe(pc.STATE.CERTIFIED_CALIBRATED);
|
|
expect(calibrateSpy).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it('support comes from the artifact, so a nightly refit cannot widen it', async () => {
|
|
const built = await svc.build({}, { calibrationService: {
|
|
fromLedger: async () => ({ map: { x: [0, 1], y: [0.5, 0.9] }, fit_n: 900, fitted_through: 'd9',
|
|
bands: [[0.0, 1.0]] }) } }); // fit claims the whole range
|
|
expect(built.resolve({ model_version: ERA, p_win: 0.95 }).probability_state).toBe(pc.STATE.UNCERTIFIED);
|
|
});
|
|
});
|