22cf51c4b0
Runtime probes say the fleet is on a8de676, one process generation. But
"isotonic" was a label, not a claim: probabilityContractService refits per
snapshot against `game_date < todayEt()`, so the mapping changes as outcomes
settle, and nothing on a row could say WHICH mapping produced its number.
The artifact now has an identity:
estimator_type / estimator_version / certification_version / model_version
fit_as_of the exact lt(game_date) bound 2026-09-02
training_cutoff last date INSIDE the fit 2026-08-21
fit_n / knot_count 6,084 / 28
knot_digest d9d571d728ba76de
served_curve the COMPLETE served function over [0.50,0.80)
served_curve_digest
The served curve is not a sample. p_win is quantised to three decimals at the
source, so a step table at 0.001 granularity is the mapping itself for every
input that can occur — six steps, ~200 bytes. Storing it makes a Read
reconstructable WITHOUT re-deriving a training set that may since have been
re-settled, and a claim you can only verify when the inputs happen not to have
moved is not a reconstructable claim.
Proven, not asserted: the production construction path run twice gives an
identical digest, and an INDEPENDENT reconstruction — re-walk 9,361 settled
ledger rows at the declared bound, refit from scratch — reproduces
d9d571d728ba76de exactly, 28 knots for 28.
A teeth injection found a real defect behind a coverage hole. `resolve` checked
the CONTRACT's model era and never the ARTIFACT's, so a mapping fitted for a
different era could be recorded beside a served number with every test green.
Both the era and the estimator type are now checked, and a mismatch serves
nothing rather than serving quietly.
OBSERVED AND NOT CHANGED: calibrationService splits 65/35 to certify its own
bands, a step this contract does not consume because support comes from the
frozen artifact. So the served map is fitted through 2026-08-21 while 3,277
more recent settled rows sit unused, and that lag grows with history. Changing
it would change the fitted function, which this tranche froze.
Shadow still defaults OFF. CALIBRATION_DEPLOYED still []. served_probability is
referenced by nothing outside the contract layer — asserted by a tooth.
Suite 401/401, 5,593 passed, 4 skipped. Teeth 23/23 (prior) + 7/7 (new).
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CQJeAG8vcDoL5zkiaJyVb8
330 lines
15 KiB
JavaScript
330 lines
15 KiB
JavaScript
'use strict';
|
|
|
|
/**
|
|
* THE CONTRACT'S ONE JOB: never answer "the calibrator is not supported here"
|
|
* with a number we have already measured to be wrong.
|
|
*/
|
|
const pc = require('../../src/services/model/probabilityContract');
|
|
const svc = require('../../src/services/model/probabilityContractService');
|
|
const reg = require('../../src/services/model/calibrationRegistry');
|
|
|
|
const ERA = 'engine1@2026-08-07-fullwindow';
|
|
const read = (p, over = {}) => ({ sport: 'mlb', stat: 'hits', model_version: ERA, p_win: p, ...over });
|
|
// A stand-in shaped like the real map: monotone, flattening, correcting downward.
|
|
const iso = (p) => Math.round((0.42 + 0.26 * p) * 1000) / 1000;
|
|
const deps = { estimate: iso };
|
|
|
|
describe('probability object and states', () => {
|
|
it('serves a calibrated number inside certified raw support', () => {
|
|
const r = pc.resolve(read(0.65), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.CERTIFIED_CALIBRATED);
|
|
expect(r.served_probability).toBe(iso(0.65));
|
|
expect(pc.isCertified(r)).toBe(true);
|
|
});
|
|
|
|
it('NO RAW FALLBACK — an uncertified region serves no number at all', () => {
|
|
for (const p of [0.85, 0.90, 0.95, 0.99]) {
|
|
const r = pc.resolve(read(p), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.UNCERTIFIED);
|
|
expect(r.served_probability).toBeNull();
|
|
// the specific defect: served must not silently become the raw value
|
|
expect(r.served_probability).not.toBe(p);
|
|
}
|
|
});
|
|
|
|
it('raw probability is preserved in every state, including refusals', () => {
|
|
for (const p of [0.45, 0.55, 0.85, 0.95]) {
|
|
expect(pc.resolve(read(p), deps).raw_model_probability).toBe(p);
|
|
}
|
|
expect(pc.resolve(read(0.9), deps).raw_model_probability).toBe(0.9);
|
|
});
|
|
|
|
it('a different model era is a different forecaster — VERSION_MISMATCH, no number', () => {
|
|
const r = pc.resolve(read(0.65, { model_version: 'engine1@2026-07-20' }), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.VERSION_MISMATCH);
|
|
expect(r.served_probability).toBeNull();
|
|
});
|
|
|
|
it('another stat or sport is UNSUPPORTED, never quietly served', () => {
|
|
for (const o of [{ stat: 'total_bases' }, { stat: 'rbi' }, { sport: 'wnba' }, { sport: 'nba' }]) {
|
|
const r = pc.resolve(read(0.65, o), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.UNSUPPORTED);
|
|
expect(r.served_probability).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('an absent or out-of-range raw probability is INVALID, not coerced', () => {
|
|
for (const p of [null, undefined, NaN, -0.1, 1.4, 'x']) {
|
|
const r = pc.resolve(read(p), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.INVALID);
|
|
expect(r.served_probability).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('refuses when the estimator declines inside its own support', () => {
|
|
const r = pc.resolve(read(0.65), { estimate: () => null });
|
|
expect(r.probability_state).toBe(pc.STATE.UNCERTIFIED);
|
|
expect(r.served_probability).toBeNull();
|
|
});
|
|
|
|
it('the certified band is half-open and expressed in the RAW input domain', () => {
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.50)).toBe(true);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.799)).toBe(true);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.80)).toBe(false);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.499)).toBe(false);
|
|
});
|
|
});
|
|
|
|
describe('the served function is non-decreasing across its support', () => {
|
|
it('never moves backwards as raw confidence rises', () => {
|
|
let prev = null;
|
|
for (let x = 0.30; x <= 1.0001; x += 0.001) {
|
|
const p = Math.round(x * 1000) / 1000;
|
|
const r = pc.resolve(read(p), deps);
|
|
if (r.served_probability == null) continue;
|
|
if (prev != null) expect(r.served_probability).toBeGreaterThanOrEqual(prev);
|
|
prev = r.served_probability;
|
|
}
|
|
});
|
|
|
|
it('a gap is an absence, not a step down to raw', () => {
|
|
const inside = pc.resolve(read(0.799), deps).served_probability;
|
|
const outside = pc.resolve(read(0.80), deps);
|
|
expect(inside).not.toBeNull();
|
|
expect(outside.served_probability).toBeNull();
|
|
expect(outside.served_probability).not.toBe(0.80);
|
|
});
|
|
});
|
|
|
|
describe('actionability law', () => {
|
|
const odds = -115;
|
|
it('derives EV, Kelly and VALUE from the SERVED probability', () => {
|
|
const r = pc.resolve(read(0.65), deps);
|
|
const d = pc.derivedClaims(r, odds);
|
|
expect(d.available).toBe(true);
|
|
expect(d.computed_from).toBe('served_probability');
|
|
const { evPct } = require('../../src/utils/devig');
|
|
expect(d.ev_pct).toBe(evPct(r.served_probability, odds));
|
|
expect(d.ev_pct).not.toBe(evPct(0.65, odds)); // NOT from raw
|
|
});
|
|
|
|
it('withdraws EV, Kelly and VALUE entirely when no probability is certified', () => {
|
|
for (const p of [0.85, 0.95]) {
|
|
const d = pc.derivedClaims(pc.resolve(read(p), deps), odds);
|
|
expect(d.available).toBe(false);
|
|
expect(d.ev_pct).toBeNull();
|
|
expect(d.kelly).toBeNull();
|
|
expect(d.value).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('never computes a derived claim from raw behind the scenes', () => {
|
|
const { evPct } = require('../../src/utils/devig');
|
|
const { quarterKelly } = require('../../src/utils/kelly');
|
|
const d = pc.derivedClaims(pc.resolve(read(0.91), deps), odds);
|
|
expect(d.ev_pct).not.toBe(evPct(0.91, odds));
|
|
expect(d.kelly).not.toEqual(quarterKelly(0.91, odds));
|
|
expect(d.kelly).toBeNull();
|
|
});
|
|
|
|
it('a VERSION_MISMATCH withdraws actionability too', () => {
|
|
const d = pc.derivedClaims(pc.resolve(read(0.65, { model_version: 'other' }), deps), odds);
|
|
expect(d.available).toBe(false);
|
|
});
|
|
});
|
|
|
|
describe('confidence display', () => {
|
|
it('shows an exact number only when the state is certified', () => {
|
|
const c = pc.confidenceDisplay(pc.resolve(read(0.65), deps));
|
|
expect(c.exact_probability).toBe(iso(0.65));
|
|
expect(c.calibrated).toBe(true);
|
|
});
|
|
|
|
it('shows NO exact confidence when uncertified — and never the raw value', () => {
|
|
const c = pc.confidenceDisplay(pc.resolve(read(0.91), deps));
|
|
expect(c.exact_probability).toBeNull();
|
|
expect(c.exact_pct).toBeNull();
|
|
expect(c.calibrated).toBe(false);
|
|
expect(c.label).toBe('Confidence not calibrated');
|
|
expect(JSON.stringify(c)).not.toContain('0.91');
|
|
});
|
|
});
|
|
|
|
describe('the artifact records its own adjudication', () => {
|
|
it('is pinned to the current model era and the isotonic estimator', () => {
|
|
expect(pc.MLB_HITS.model_version).toBe(ERA);
|
|
expect(pc.MLB_HITS.estimator_type).toBe(pc.ESTIMATOR.ISOTONIC);
|
|
expect(pc.MLB_HITS.estimator_version).toBeTruthy();
|
|
expect(pc.MLB_HITS.certification_version).toBeTruthy();
|
|
});
|
|
|
|
it('certifies nothing above raw 0.80 — the region where raw is most wrong', () => {
|
|
for (const p of [0.80, 0.85, 0.90, 0.95]) {
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, p)).toBe(false);
|
|
}
|
|
});
|
|
|
|
it('holds ONE contract — no other sport or stat is certified', () => {
|
|
expect(Object.keys(pc.CONTRACTS)).toEqual(['mlb:hits']);
|
|
});
|
|
|
|
it('the held-out interval it records excludes zero', () => {
|
|
expect(pc.MLB_HITS.evidence.ci95[1]).toBeLessThan(0);
|
|
});
|
|
});
|
|
|
|
describe('the registry already asked the right question', () => {
|
|
it('serves() tests certified bands against the RAW p_win, not the output', () => {
|
|
const r = reg.createRegistry();
|
|
r.deploy('hits', { lodo_pass: true, ci: [-0.006, -0.001], map: { x: [0], y: [0] },
|
|
certified_bands: [[0.50, 0.80]] });
|
|
expect(r.serves('hits', 0.65).serve).toBe(true);
|
|
expect(r.serves('hits', 0.90).serve).toBe(false);
|
|
});
|
|
|
|
it('a refusal names the reason and never proposes raw', () => {
|
|
const r = reg.createRegistry();
|
|
r.deploy('hits', { lodo_pass: true, ci: [-0.006, -0.001], map: {}, certified_bands: [[0.50, 0.80]] });
|
|
const out = r.serves('hits', 0.92);
|
|
expect(out.serve).toBe(false);
|
|
expect(JSON.stringify(out)).not.toContain('0.92');
|
|
});
|
|
});
|
|
|
|
describe('probabilityContractService', () => {
|
|
it('returns null — not raw — when there is no settled history', async () => {
|
|
const built = await svc.build({}, { calibrationService: { fromLedger: async () => null } });
|
|
expect(built).toBeNull();
|
|
});
|
|
|
|
it('takes the MAP and never the blocked calibrate() gate', async () => {
|
|
const cal = require('../../src/services/model/calibration');
|
|
const map = cal.fitIsotonic(Array.from({ length: 600 }, (_, i) => {
|
|
const p = 0.40 + (i % 55) / 100;
|
|
return { p, won: i % 3 === 0 ? 0 : 1, date: `d${i % 12}` };
|
|
}), { minTotal: 200 });
|
|
const calibrateSpy = jest.fn(() => ({ p_calibrated: 0.99, calibrated: true }));
|
|
const built = await svc.build({}, { calibrationService: {
|
|
fromLedger: async () => ({ map, fit_n: 600, fitted_through: 'd11', calibrate: calibrateSpy }) } });
|
|
expect(built).not.toBeNull();
|
|
const r = built.resolve({ model_version: ERA, p_win: 0.65 });
|
|
expect(r.probability_state).toBe(pc.STATE.CERTIFIED_CALIBRATED);
|
|
expect(calibrateSpy).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it('support comes from the artifact, so a nightly refit cannot widen it', async () => {
|
|
const built = await svc.build({}, { calibrationService: {
|
|
fromLedger: async () => ({ map: { x: [0, 1], y: [0.5, 0.9] }, fit_n: 900, fitted_through: 'd9',
|
|
bands: [[0.0, 1.0]] }) } }); // fit claims the whole range
|
|
expect(built.resolve({ model_version: ERA, p_win: 0.95 }).probability_state).toBe(pc.STATE.UNCERTIFIED);
|
|
});
|
|
});
|
|
|
|
describe('artifact identity — the mapping that actually ran', () => {
|
|
const cal = require('../../src/services/model/calibration');
|
|
const mk = (seed) => cal.fitIsotonic(Array.from({ length: 900 }, (_, i) => {
|
|
const p = Math.round((0.35 + (i % 60) / 100) * 1000) / 1000;
|
|
return { p, won: ((i * seed) % 1000) / 1000 < (0.5 + 0.45 * (p - 0.5)) ? 1 : 0, date: `d${i % 14}` };
|
|
}), { minTotal: 200 });
|
|
const fitted = (map, over = {}) => ({ calibrationService: { fromLedger: async () => ({
|
|
map, fit_n: 900, fitted_through: 'd13', cutoff: '2026-09-03', ...over }) } });
|
|
|
|
it('the same evidence reconstructs the same artifact, digest for digest', async () => {
|
|
const map = mk(2654435761);
|
|
const a = await svc.build({}, fitted(map));
|
|
const b = await svc.build({}, fitted(map));
|
|
expect(a.artifact.knot_digest).toBe(b.artifact.knot_digest);
|
|
expect(a.artifact.served_curve_digest).toBe(b.artifact.served_curve_digest);
|
|
expect(a.artifact.served_curve).toEqual(b.artifact.served_curve);
|
|
});
|
|
|
|
it('DIFFERENT evidence produces a different identity — the digest is not decorative', async () => {
|
|
const a = await svc.build({}, fitted(mk(2654435761)));
|
|
const b = await svc.build({}, fitted(mk(40503)));
|
|
expect(a.artifact.knot_digest).not.toBe(b.artifact.knot_digest);
|
|
});
|
|
|
|
it('carries the point-in-time bound and the training cutoff, distinctly', async () => {
|
|
const a = await svc.build({}, fitted(mk(2654435761)));
|
|
expect(a.artifact.fit_as_of).toBe('2026-09-03'); // the lt(game_date) bound
|
|
expect(a.artifact.training_cutoff).toBe('d13'); // last date inside the fit
|
|
expect(a.artifact.fit_n).toBe(900);
|
|
expect(a.artifact.knot_count).toBeGreaterThan(0);
|
|
});
|
|
|
|
it('the served curve IS the served function over certified support, not a sample', async () => {
|
|
const a = await svc.build({}, fitted(mk(2654435761)));
|
|
const lookup = (raw) => {
|
|
let v = null;
|
|
for (const [from, val] of a.artifact.served_curve) if (raw >= from) v = val;
|
|
return v;
|
|
};
|
|
for (let x = 0.50; x < 0.80 - 1e-9; x += 0.001) {
|
|
const raw = Math.round(x * 1000) / 1000;
|
|
const r = a.resolve({ model_version: ERA, p_win: raw });
|
|
expect(r.served_probability).toBe(Math.round(lookup(raw) * 1000) / 1000);
|
|
}
|
|
});
|
|
|
|
it('the curve covers ONLY certified support — never the unsupported tail', async () => {
|
|
const a = await svc.build({}, fitted(mk(2654435761)));
|
|
for (const [from] of a.artifact.served_curve) {
|
|
expect(from).toBeGreaterThanOrEqual(0.50);
|
|
expect(from).toBeLessThan(0.80);
|
|
}
|
|
});
|
|
|
|
it('a resolution with no artifact records null rather than inventing one', () => {
|
|
const r = pc.resolve(read(0.65), { estimate: iso });
|
|
expect(r.artifact).toBeNull();
|
|
expect(r.served_probability).not.toBeNull();
|
|
});
|
|
});
|
|
|
|
describe('point in time — a Read can only see settlements before its own day', () => {
|
|
const calSvc = require('../../src/services/model/calibrationService');
|
|
|
|
/** Records the filters actually applied, so this tests behaviour not source. */
|
|
function recordingClient(rows) {
|
|
const applied = [];
|
|
const q = {
|
|
select: () => q, eq: (c, v) => { applied.push(['eq', c, v]); return q; },
|
|
is: (c, v) => { applied.push(['is', c, v]); return q; },
|
|
in: (c, v) => { applied.push(['in', c, v]); return q; },
|
|
not: (c, o, v) => { applied.push(['not', c, o, v]); return q; },
|
|
lt: (c, v) => { applied.push(['lt', c, v]); return q; },
|
|
lte: (c, v) => { applied.push(['lte', c, v]); return q; },
|
|
gte: (c, v) => { applied.push(['gte', c, v]); return q; },
|
|
order: () => q, limit: () => q,
|
|
range: async () => ({ data: rows, error: null, count: rows.length }),
|
|
then: (res) => res({ data: rows, error: null }),
|
|
};
|
|
return { applied, client: { from: () => q } };
|
|
}
|
|
|
|
it('bounds the training walk STRICTLY BEFORE the cutoff, never at or after it', async () => {
|
|
const { applied, client } = recordingClient([]);
|
|
await calSvc.loadSettledRows(client, { sport: 'mlb', stat: 'hits', before: '2026-09-02' });
|
|
const bound = applied.filter((a) => a[1] === 'game_date');
|
|
expect(bound.length).toBeGreaterThan(0);
|
|
// strictly-less is the whole discipline: `lte` would admit same-day games
|
|
expect(bound.some((a) => a[0] === 'lt' && a[2] === '2026-09-02')).toBe(true);
|
|
expect(bound.some((a) => a[0] === 'lte')).toBe(false);
|
|
expect(bound.some((a) => a[0] === 'gte')).toBe(false);
|
|
});
|
|
|
|
it('the artifact records the bound it was fitted under, so a Read names its own evidence horizon', async () => {
|
|
const cal = require('../../src/services/model/calibration');
|
|
const map = cal.fitIsotonic(Array.from({ length: 900 }, (_, i) => {
|
|
const p = Math.round((0.35 + (i % 60) / 100) * 1000) / 1000;
|
|
return { p, won: ((i * 2654435761) % 1000) / 1000 < (0.5 + 0.45 * (p - 0.5)) ? 1 : 0, date: `d${i % 14}` };
|
|
}), { minTotal: 200 });
|
|
const built = await svc.build({}, { calibrationService: { fromLedger: async () => ({
|
|
map, fit_n: 900, fitted_through: '2026-08-21', cutoff: '2026-09-02' }) } });
|
|
expect(built.artifact.fit_as_of).toBe('2026-09-02');
|
|
expect(built.artifact.training_cutoff).toBe('2026-08-21');
|
|
// and the two are DIFFERENT questions — the bound, and the last date inside it
|
|
expect(built.artifact.fit_as_of).not.toBe(built.artifact.training_cutoff);
|
|
});
|
|
});
|