94f7c3c3ef
Post-fix cohort df4ec562 closed the primary question: eleven non-hits stats, 1,910 rows, ZERO certified and ZERO numeric served probabilities. The cross-stat repair holds. But every one of those 1,910 rows still recorded `artifact_id: mlb-hits-isotonic@2026-09-03` beside state UNSUPPORTED. A `doubles` row named the hits artifact. Nothing was calibrated by it, so no number leaked — but a later query for "rows this artifact produced" would have returned 2,186 instead of 133, and that is the shape of footgun this programme keeps finding. The contract check runs BEFORE any artifact is relevant: with no certified contract for the sport/stat, no artifact applies, and naming one asserts a relationship that does not exist. UNSUPPORTED now carries null artifact, artifact_id, estimator_type, estimator_version, certification_version and procedure_version. Attribution is KEPT where the artifact is genuinely the thing that declined — UNCERTIFIED (out of support) and VERSION_MISMATCH both still name it. A test holds both directions so this does not over-correct into erasing real provenance. One existing test called resolve() without naming a stat and relied on the service substituting one. That substitution was the original defect, so the test now names its stat, as production does. Artifact unchanged. Live OFF. Suite 405/405, 5,662 passed. Teeth 35/35 + 10/10 + 23/23. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CQJeAG8vcDoL5zkiaJyVb8
317 lines
13 KiB
JavaScript
317 lines
13 KiB
JavaScript
'use strict';
|
|
|
|
/**
|
|
* THE CONTRACT'S ONE JOB: never answer "the calibrator is not supported here"
|
|
* with a number we have already measured to be wrong.
|
|
*/
|
|
const pc = require('../../src/services/model/probabilityContract');
|
|
const svc = require('../../src/services/model/probabilityContractService');
|
|
const reg = require('../../src/services/model/calibrationRegistry');
|
|
|
|
const ERA = 'engine1@2026-08-07-fullwindow';
|
|
const read = (p, over = {}) => ({ sport: 'mlb', stat: 'hits', model_version: ERA, p_win: p, ...over });
|
|
// A stand-in shaped like the real map: monotone, flattening, correcting downward.
|
|
const iso = (p) => Math.round((0.42 + 0.26 * p) * 1000) / 1000;
|
|
const deps = { estimate: iso };
|
|
|
|
describe('probability object and states', () => {
|
|
it('serves a calibrated number inside certified raw support', () => {
|
|
const r = pc.resolve(read(0.65), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.CERTIFIED_CALIBRATED);
|
|
expect(r.served_probability).toBe(iso(0.65));
|
|
expect(pc.isCertified(r)).toBe(true);
|
|
});
|
|
|
|
it('NO RAW FALLBACK — an uncertified region serves no number at all', () => {
|
|
for (const p of [0.85, 0.90, 0.95, 0.99]) {
|
|
const r = pc.resolve(read(p), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.UNCERTIFIED);
|
|
expect(r.served_probability).toBeNull();
|
|
// the specific defect: served must not silently become the raw value
|
|
expect(r.served_probability).not.toBe(p);
|
|
}
|
|
});
|
|
|
|
it('raw probability is preserved in every state, including refusals', () => {
|
|
for (const p of [0.45, 0.55, 0.85, 0.95]) {
|
|
expect(pc.resolve(read(p), deps).raw_model_probability).toBe(p);
|
|
}
|
|
expect(pc.resolve(read(0.9), deps).raw_model_probability).toBe(0.9);
|
|
});
|
|
|
|
it('a different model era is a different forecaster — VERSION_MISMATCH, no number', () => {
|
|
const r = pc.resolve(read(0.65, { model_version: 'engine1@2026-07-20' }), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.VERSION_MISMATCH);
|
|
expect(r.served_probability).toBeNull();
|
|
});
|
|
|
|
it('another stat or sport is UNSUPPORTED, never quietly served', () => {
|
|
for (const o of [{ stat: 'total_bases' }, { stat: 'rbi' }, { sport: 'wnba' }, { sport: 'nba' }]) {
|
|
const r = pc.resolve(read(0.65, o), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.UNSUPPORTED);
|
|
expect(r.served_probability).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('an absent or out-of-range raw probability is INVALID, not coerced', () => {
|
|
for (const p of [null, undefined, NaN, -0.1, 1.4, 'x']) {
|
|
const r = pc.resolve(read(p), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.INVALID);
|
|
expect(r.served_probability).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('refuses when the estimator declines inside its own support', () => {
|
|
const r = pc.resolve(read(0.65), { estimate: () => null });
|
|
expect(r.probability_state).toBe(pc.STATE.UNCERTIFIED);
|
|
expect(r.served_probability).toBeNull();
|
|
});
|
|
|
|
it('the certified band is half-open and expressed in the RAW input domain', () => {
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.50)).toBe(true);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.799)).toBe(true);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.80)).toBe(false);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.499)).toBe(false);
|
|
});
|
|
});
|
|
|
|
describe('the served function is non-decreasing across its support', () => {
|
|
it('never moves backwards as raw confidence rises', () => {
|
|
let prev = null;
|
|
for (let x = 0.30; x <= 1.0001; x += 0.001) {
|
|
const p = Math.round(x * 1000) / 1000;
|
|
const r = pc.resolve(read(p), deps);
|
|
if (r.served_probability == null) continue;
|
|
if (prev != null) expect(r.served_probability).toBeGreaterThanOrEqual(prev);
|
|
prev = r.served_probability;
|
|
}
|
|
});
|
|
|
|
it('a gap is an absence, not a step down to raw', () => {
|
|
const inside = pc.resolve(read(0.799), deps).served_probability;
|
|
const outside = pc.resolve(read(0.80), deps);
|
|
expect(inside).not.toBeNull();
|
|
expect(outside.served_probability).toBeNull();
|
|
expect(outside.served_probability).not.toBe(0.80);
|
|
});
|
|
});
|
|
|
|
describe('actionability law', () => {
|
|
const odds = -115;
|
|
it('derives EV, Kelly and VALUE from the SERVED probability', () => {
|
|
const r = pc.resolve(read(0.65), deps);
|
|
const d = pc.derivedClaims(r, odds);
|
|
expect(d.available).toBe(true);
|
|
expect(d.computed_from).toBe('served_probability');
|
|
const { evPct } = require('../../src/utils/devig');
|
|
expect(d.ev_pct).toBe(evPct(r.served_probability, odds));
|
|
expect(d.ev_pct).not.toBe(evPct(0.65, odds)); // NOT from raw
|
|
});
|
|
|
|
it('withdraws EV, Kelly and VALUE entirely when no probability is certified', () => {
|
|
for (const p of [0.85, 0.95]) {
|
|
const d = pc.derivedClaims(pc.resolve(read(p), deps), odds);
|
|
expect(d.available).toBe(false);
|
|
expect(d.ev_pct).toBeNull();
|
|
expect(d.kelly).toBeNull();
|
|
expect(d.value).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('never computes a derived claim from raw behind the scenes', () => {
|
|
const { evPct } = require('../../src/utils/devig');
|
|
const { quarterKelly } = require('../../src/utils/kelly');
|
|
const d = pc.derivedClaims(pc.resolve(read(0.91), deps), odds);
|
|
expect(d.ev_pct).not.toBe(evPct(0.91, odds));
|
|
expect(d.kelly).not.toEqual(quarterKelly(0.91, odds));
|
|
expect(d.kelly).toBeNull();
|
|
});
|
|
|
|
it('a VERSION_MISMATCH withdraws actionability too', () => {
|
|
const d = pc.derivedClaims(pc.resolve(read(0.65, { model_version: 'other' }), deps), odds);
|
|
expect(d.available).toBe(false);
|
|
});
|
|
});
|
|
|
|
describe('confidence display', () => {
|
|
it('shows an exact number only when the state is certified', () => {
|
|
const c = pc.confidenceDisplay(pc.resolve(read(0.65), deps));
|
|
expect(c.exact_probability).toBe(iso(0.65));
|
|
expect(c.calibrated).toBe(true);
|
|
});
|
|
|
|
it('shows NO exact confidence when uncertified — and never the raw value', () => {
|
|
const c = pc.confidenceDisplay(pc.resolve(read(0.91), deps));
|
|
expect(c.exact_probability).toBeNull();
|
|
expect(c.exact_pct).toBeNull();
|
|
expect(c.calibrated).toBe(false);
|
|
expect(c.label).toBe('Confidence not calibrated');
|
|
expect(JSON.stringify(c)).not.toContain('0.91');
|
|
});
|
|
});
|
|
|
|
describe('the artifact records its own adjudication', () => {
|
|
it('is pinned to the current model era and the isotonic estimator', () => {
|
|
expect(pc.MLB_HITS.model_version).toBe(ERA);
|
|
expect(pc.MLB_HITS.estimator_type).toBe(pc.ESTIMATOR.ISOTONIC);
|
|
expect(pc.MLB_HITS.estimator_version).toBeTruthy();
|
|
expect(pc.MLB_HITS.certification_version).toBeTruthy();
|
|
});
|
|
|
|
it('certifies nothing above raw 0.80 — the region where raw is most wrong', () => {
|
|
for (const p of [0.80, 0.85, 0.90, 0.95]) {
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, p)).toBe(false);
|
|
}
|
|
});
|
|
|
|
it('holds ONE contract — no other sport or stat is certified', () => {
|
|
expect(Object.keys(pc.CONTRACTS)).toEqual(['mlb:hits']);
|
|
});
|
|
|
|
it('the held-out interval it records excludes zero', () => {
|
|
expect(pc.MLB_HITS.evidence.ci95[1]).toBeLessThan(0);
|
|
});
|
|
});
|
|
|
|
describe('the registry already asked the right question', () => {
|
|
it('serves() tests certified bands against the RAW p_win, not the output', () => {
|
|
const r = reg.createRegistry();
|
|
r.deploy('hits', { lodo_pass: true, ci: [-0.006, -0.001], map: { x: [0], y: [0] },
|
|
certified_bands: [[0.50, 0.80]] });
|
|
expect(r.serves('hits', 0.65).serve).toBe(true);
|
|
expect(r.serves('hits', 0.90).serve).toBe(false);
|
|
});
|
|
|
|
it('a refusal names the reason and never proposes raw', () => {
|
|
const r = reg.createRegistry();
|
|
r.deploy('hits', { lodo_pass: true, ci: [-0.006, -0.001], map: {}, certified_bands: [[0.50, 0.80]] });
|
|
const out = r.serves('hits', 0.92);
|
|
expect(out.serve).toBe(false);
|
|
expect(JSON.stringify(out)).not.toContain('0.92');
|
|
});
|
|
});
|
|
|
|
describe('probabilityContractService — loads, never fits', () => {
|
|
const registry = require('../../src/services/model/artifactRegistry');
|
|
|
|
it('returns the PROMOTED frozen artifact, not a fresh fit', async () => {
|
|
const b = await svc.build(null, { sport: 'mlb', stat: 'hits' });
|
|
expect(b).not.toBeNull();
|
|
expect(b.artifact.artifact_id).toBe(registry.PROMOTED['mlb:hits'].artifact_id);
|
|
expect(b.artifact.servable).toBe(true);
|
|
});
|
|
|
|
it('returns null — never a fallback curve — when nothing is promoted', async () => {
|
|
for (const stat of ['rbi', 'total_bases', 'runs']) {
|
|
expect(await svc.build(null, { sport: 'mlb', stat })).toBeNull();
|
|
}
|
|
expect(await svc.build(null, { sport: 'wnba', stat: 'points' })).toBeNull();
|
|
});
|
|
|
|
it('needs no database client at all — there is nothing left to read', async () => {
|
|
const b = await svc.build(undefined, { sport: 'mlb', stat: 'hits' });
|
|
expect(b.artifact.artifact_id).toBeTruthy();
|
|
});
|
|
|
|
it('two resolutions of the same input are identical, forever', async () => {
|
|
const b1 = await svc.build(null, { sport: 'mlb', stat: 'hits' });
|
|
const b2 = await svc.build(null, { sport: 'mlb', stat: 'hits' });
|
|
for (const p of [0.50, 0.55, 0.601, 0.72, 0.799]) {
|
|
// the read NAMES its stat — the resolver no longer substitutes one
|
|
const a = b1.resolve({ sport: 'mlb', stat: 'hits', model_version: ERA, p_win: p });
|
|
const c = b2.resolve({ sport: 'mlb', stat: 'hits', model_version: ERA, p_win: p });
|
|
expect(a.served_probability).toBe(c.served_probability);
|
|
expect(a.artifact.knot_digest).toBe(c.artifact.knot_digest);
|
|
}
|
|
});
|
|
});
|
|
|
|
describe('artifact identity — the frozen curve that actually ran', () => {
|
|
const registry = require('../../src/services/model/artifactRegistry');
|
|
const a = registry.load('mlb', 'hits');
|
|
|
|
it('carries every field needed to find and check it again', () => {
|
|
for (const k of ['artifact_id', 'procedure_version', 'sport', 'stat', 'model_version',
|
|
'fit_as_of', 'training_cutoff', 'fit_n', 'source_digest', 'algorithm', 'algorithm_version',
|
|
'knot_count', 'knot_digest', 'served_curve', 'served_curve_digest', 'certified_bands']) {
|
|
expect(a[k]).toBeDefined();
|
|
expect(a[k]).not.toBeNull();
|
|
}
|
|
});
|
|
|
|
it('fit_as_of and training_cutoff are DIFFERENT questions', () => {
|
|
// the eligibility bound, and the newest observation actually admitted
|
|
expect(a.fit_as_of).not.toBe(a.training_cutoff);
|
|
expect(a.training_cutoff < a.fit_as_of).toBe(true);
|
|
});
|
|
|
|
it('was fitted on the current era ONLY', () => {
|
|
expect(a.model_version).toBe(ERA);
|
|
expect(a.era_audit.wrong_era_rows).toBe(0);
|
|
expect(Object.keys(a.era_audit.era_counts)).toEqual([ERA]);
|
|
});
|
|
|
|
it('withheld nothing from the promoted fit', () => {
|
|
expect(a.withheld_from_fit).toBe(0);
|
|
expect(a.fit_n).toBe(a.era_audit.current_era_rows);
|
|
});
|
|
|
|
it('the served curve IS the served function over support, not a sample', () => {
|
|
for (let x = 0.50; x < 0.80 - 1e-9; x += 0.001) {
|
|
const raw = Math.round(x * 1000) / 1000;
|
|
let want = null;
|
|
for (const [from, val] of a.served_curve) { if (raw >= from) want = val; else break; }
|
|
expect(registry.applyCurve(a, raw)).toBe(want);
|
|
}
|
|
});
|
|
|
|
it('the curve covers ONLY certified support', () => {
|
|
for (const [from] of a.served_curve) {
|
|
expect(from).toBeGreaterThanOrEqual(0.50);
|
|
expect(from).toBeLessThan(0.80);
|
|
}
|
|
for (const p of [0.499, 0.80, 0.9, 0.99]) expect(registry.applyCurve(a, p)).toBeNull();
|
|
});
|
|
});
|
|
|
|
describe('point in time — a Read can only see settlements before its own day', () => {
|
|
const calSvc = require('../../src/services/model/calibrationService');
|
|
|
|
/** Records the filters actually applied, so this tests behaviour not source. */
|
|
function recordingClient(rows) {
|
|
const applied = [];
|
|
const q = {
|
|
select: () => q, eq: (c, v) => { applied.push(['eq', c, v]); return q; },
|
|
is: (c, v) => { applied.push(['is', c, v]); return q; },
|
|
in: (c, v) => { applied.push(['in', c, v]); return q; },
|
|
not: (c, o, v) => { applied.push(['not', c, o, v]); return q; },
|
|
lt: (c, v) => { applied.push(['lt', c, v]); return q; },
|
|
lte: (c, v) => { applied.push(['lte', c, v]); return q; },
|
|
gte: (c, v) => { applied.push(['gte', c, v]); return q; },
|
|
order: () => q, limit: () => q,
|
|
range: async () => ({ data: rows, error: null, count: rows.length }),
|
|
then: (res) => res({ data: rows, error: null }),
|
|
};
|
|
return { applied, client: { from: () => q } };
|
|
}
|
|
|
|
it('bounds the training walk STRICTLY BEFORE the cutoff, never at or after it', async () => {
|
|
const { applied, client } = recordingClient([]);
|
|
await calSvc.loadSettledRows(client, { sport: 'mlb', stat: 'hits', before: '2026-09-02' });
|
|
const bound = applied.filter((a) => a[1] === 'game_date');
|
|
expect(bound.length).toBeGreaterThan(0);
|
|
// strictly-less is the whole discipline: `lte` would admit same-day games
|
|
expect(bound.some((a) => a[0] === 'lt' && a[2] === '2026-09-02')).toBe(true);
|
|
expect(bound.some((a) => a[0] === 'lte')).toBe(false);
|
|
expect(bound.some((a) => a[0] === 'gte')).toBe(false);
|
|
});
|
|
|
|
it('the artifact names its own evidence horizon', async () => {
|
|
const built = await svc.build(null, { sport: 'mlb', stat: 'hits' });
|
|
expect(built.artifact.fit_as_of).toBeTruthy();
|
|
expect(built.artifact.training_cutoff).toBeTruthy();
|
|
// the bound, and the last date inside it — never the same claim
|
|
expect(built.artifact.training_cutoff < built.artifact.fit_as_of).toBe(true);
|
|
});
|
|
});
|