be8e16aca9
The last release detected the violation and then served the certified state anyway. A validator that changes nothing is decoration, so `servable:false` is now load-bearing: an artifact that fails its policy returns ARTIFACT_POLICY_BLOCKED with no number, and every probability-derived claim goes with it. The gate sits inside the resolution, not beside the flag that turns the shadow on, so no environment variable can reach past it — a test asserts `resolve` never reads process.env at all. Shadow and live consume the SAME decision, differing only in which promotion stage they demand. Era mismatch still resolves to VERSION_MISMATCH rather than the new state. "This artifact belongs to a different forecaster" is more precise than "policy blocked", and the existing state already says it exactly. THE REPAIR. `currentEraSource` filters on model_version in the QUERY, taking the era from config/modelVersion so the query, the artifact and the validator all read one identity. Measured on the actual fitted set, not a second count: 6,069 current-era rows, 0 wrong-era. The procedure was then certified on current-era rows ONLY — four walk-forward folds, training strictly before each evaluation block, 0 future rows in train on every fold. All four improve; pooled n=3,108 gives Brier 0.24701 -> 0.24323, delta -0.00378, CI [-0.00619,-0.00147] excluding zero; ECE falls in every fold. Mapping spread inside support is 0.001-0.018. The prior mixed-era certification did not substitute for this. Policy B selected. A (era-filtered 65/35) and B (all current-era) are statistically indistinguishable, A-B = +0.0001 CI [-0.00029,+0.00048], but B has the better ECE (0.0064 vs 0.0109) and the holdout existed to certify the PROCEDURE — it is not permanently withheld from the artifact that ships. withheld_from_fit is 0. FROZEN. `mlb-hits-isotonic@2026-09-03`: 6,069 rows, training_cutoff 2026-09-01 (distinct from fit_as_of 2026-09-03 — the newest observation admitted is not the eligibility bound), 12 knots, source_digest 25919c16…, knot_digest 5ae940ea…, served_curve_digest c24a9dc5…, 8 curve steps, 924 bytes, committed as JSON. The runtime no longer fits. It loads. A test greps the service for fitIsotonic, fromLedger and loadRows and requires all three absent, because the old behaviour meant a user's number could move with no version, no review and no rollback, and a past Read could not be reconstructed because its curve no longer existed. New settled outcomes are forward evidence now; they cannot touch this curve. Independent reconstruction from the declared training contract alone — fresh read, fresh digest, fresh fit — reproduces every digest and the curve byte for byte. Calling the builder twice would only have proven the builder deterministic. Promotion is a frozen source constant. A snapshot cannot promote, a settlement cannot promote, a successful fit cannot promote, and dropping a file into the artifacts directory promotes nothing. Stage is APPROVED_FOR_SHADOW; live is explicitly false. Two coverage holes found by their own teeth. The promotion guard could be deleted with every test still green, because the promoted file naturally agrees with itself — extracted as `acceptFile` and tested on the case `load()` cannot reach. And `validate(null)` returned no `servable` field at all, which is falsy at a call site and so would have read as correct while asserting nothing. Shadow OFF. Live OFF. CALIBRATION_DEPLOYED []. No frontend change. Suite 404/404, 5,634 passed, 4 skipped. Teeth 26/26 + 10/10 + 23/23. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CQJeAG8vcDoL5zkiaJyVb8
316 lines
13 KiB
JavaScript
316 lines
13 KiB
JavaScript
'use strict';
|
|
|
|
/**
|
|
* THE CONTRACT'S ONE JOB: never answer "the calibrator is not supported here"
|
|
* with a number we have already measured to be wrong.
|
|
*/
|
|
const pc = require('../../src/services/model/probabilityContract');
|
|
const svc = require('../../src/services/model/probabilityContractService');
|
|
const reg = require('../../src/services/model/calibrationRegistry');
|
|
|
|
const ERA = 'engine1@2026-08-07-fullwindow';
|
|
const read = (p, over = {}) => ({ sport: 'mlb', stat: 'hits', model_version: ERA, p_win: p, ...over });
|
|
// A stand-in shaped like the real map: monotone, flattening, correcting downward.
|
|
const iso = (p) => Math.round((0.42 + 0.26 * p) * 1000) / 1000;
|
|
const deps = { estimate: iso };
|
|
|
|
describe('probability object and states', () => {
|
|
it('serves a calibrated number inside certified raw support', () => {
|
|
const r = pc.resolve(read(0.65), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.CERTIFIED_CALIBRATED);
|
|
expect(r.served_probability).toBe(iso(0.65));
|
|
expect(pc.isCertified(r)).toBe(true);
|
|
});
|
|
|
|
it('NO RAW FALLBACK — an uncertified region serves no number at all', () => {
|
|
for (const p of [0.85, 0.90, 0.95, 0.99]) {
|
|
const r = pc.resolve(read(p), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.UNCERTIFIED);
|
|
expect(r.served_probability).toBeNull();
|
|
// the specific defect: served must not silently become the raw value
|
|
expect(r.served_probability).not.toBe(p);
|
|
}
|
|
});
|
|
|
|
it('raw probability is preserved in every state, including refusals', () => {
|
|
for (const p of [0.45, 0.55, 0.85, 0.95]) {
|
|
expect(pc.resolve(read(p), deps).raw_model_probability).toBe(p);
|
|
}
|
|
expect(pc.resolve(read(0.9), deps).raw_model_probability).toBe(0.9);
|
|
});
|
|
|
|
it('a different model era is a different forecaster — VERSION_MISMATCH, no number', () => {
|
|
const r = pc.resolve(read(0.65, { model_version: 'engine1@2026-07-20' }), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.VERSION_MISMATCH);
|
|
expect(r.served_probability).toBeNull();
|
|
});
|
|
|
|
it('another stat or sport is UNSUPPORTED, never quietly served', () => {
|
|
for (const o of [{ stat: 'total_bases' }, { stat: 'rbi' }, { sport: 'wnba' }, { sport: 'nba' }]) {
|
|
const r = pc.resolve(read(0.65, o), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.UNSUPPORTED);
|
|
expect(r.served_probability).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('an absent or out-of-range raw probability is INVALID, not coerced', () => {
|
|
for (const p of [null, undefined, NaN, -0.1, 1.4, 'x']) {
|
|
const r = pc.resolve(read(p), deps);
|
|
expect(r.probability_state).toBe(pc.STATE.INVALID);
|
|
expect(r.served_probability).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('refuses when the estimator declines inside its own support', () => {
|
|
const r = pc.resolve(read(0.65), { estimate: () => null });
|
|
expect(r.probability_state).toBe(pc.STATE.UNCERTIFIED);
|
|
expect(r.served_probability).toBeNull();
|
|
});
|
|
|
|
it('the certified band is half-open and expressed in the RAW input domain', () => {
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.50)).toBe(true);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.799)).toBe(true);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.80)).toBe(false);
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, 0.499)).toBe(false);
|
|
});
|
|
});
|
|
|
|
describe('the served function is non-decreasing across its support', () => {
|
|
it('never moves backwards as raw confidence rises', () => {
|
|
let prev = null;
|
|
for (let x = 0.30; x <= 1.0001; x += 0.001) {
|
|
const p = Math.round(x * 1000) / 1000;
|
|
const r = pc.resolve(read(p), deps);
|
|
if (r.served_probability == null) continue;
|
|
if (prev != null) expect(r.served_probability).toBeGreaterThanOrEqual(prev);
|
|
prev = r.served_probability;
|
|
}
|
|
});
|
|
|
|
it('a gap is an absence, not a step down to raw', () => {
|
|
const inside = pc.resolve(read(0.799), deps).served_probability;
|
|
const outside = pc.resolve(read(0.80), deps);
|
|
expect(inside).not.toBeNull();
|
|
expect(outside.served_probability).toBeNull();
|
|
expect(outside.served_probability).not.toBe(0.80);
|
|
});
|
|
});
|
|
|
|
describe('actionability law', () => {
|
|
const odds = -115;
|
|
it('derives EV, Kelly and VALUE from the SERVED probability', () => {
|
|
const r = pc.resolve(read(0.65), deps);
|
|
const d = pc.derivedClaims(r, odds);
|
|
expect(d.available).toBe(true);
|
|
expect(d.computed_from).toBe('served_probability');
|
|
const { evPct } = require('../../src/utils/devig');
|
|
expect(d.ev_pct).toBe(evPct(r.served_probability, odds));
|
|
expect(d.ev_pct).not.toBe(evPct(0.65, odds)); // NOT from raw
|
|
});
|
|
|
|
it('withdraws EV, Kelly and VALUE entirely when no probability is certified', () => {
|
|
for (const p of [0.85, 0.95]) {
|
|
const d = pc.derivedClaims(pc.resolve(read(p), deps), odds);
|
|
expect(d.available).toBe(false);
|
|
expect(d.ev_pct).toBeNull();
|
|
expect(d.kelly).toBeNull();
|
|
expect(d.value).toBeNull();
|
|
}
|
|
});
|
|
|
|
it('never computes a derived claim from raw behind the scenes', () => {
|
|
const { evPct } = require('../../src/utils/devig');
|
|
const { quarterKelly } = require('../../src/utils/kelly');
|
|
const d = pc.derivedClaims(pc.resolve(read(0.91), deps), odds);
|
|
expect(d.ev_pct).not.toBe(evPct(0.91, odds));
|
|
expect(d.kelly).not.toEqual(quarterKelly(0.91, odds));
|
|
expect(d.kelly).toBeNull();
|
|
});
|
|
|
|
it('a VERSION_MISMATCH withdraws actionability too', () => {
|
|
const d = pc.derivedClaims(pc.resolve(read(0.65, { model_version: 'other' }), deps), odds);
|
|
expect(d.available).toBe(false);
|
|
});
|
|
});
|
|
|
|
describe('confidence display', () => {
|
|
it('shows an exact number only when the state is certified', () => {
|
|
const c = pc.confidenceDisplay(pc.resolve(read(0.65), deps));
|
|
expect(c.exact_probability).toBe(iso(0.65));
|
|
expect(c.calibrated).toBe(true);
|
|
});
|
|
|
|
it('shows NO exact confidence when uncertified — and never the raw value', () => {
|
|
const c = pc.confidenceDisplay(pc.resolve(read(0.91), deps));
|
|
expect(c.exact_probability).toBeNull();
|
|
expect(c.exact_pct).toBeNull();
|
|
expect(c.calibrated).toBe(false);
|
|
expect(c.label).toBe('Confidence not calibrated');
|
|
expect(JSON.stringify(c)).not.toContain('0.91');
|
|
});
|
|
});
|
|
|
|
describe('the artifact records its own adjudication', () => {
|
|
it('is pinned to the current model era and the isotonic estimator', () => {
|
|
expect(pc.MLB_HITS.model_version).toBe(ERA);
|
|
expect(pc.MLB_HITS.estimator_type).toBe(pc.ESTIMATOR.ISOTONIC);
|
|
expect(pc.MLB_HITS.estimator_version).toBeTruthy();
|
|
expect(pc.MLB_HITS.certification_version).toBeTruthy();
|
|
});
|
|
|
|
it('certifies nothing above raw 0.80 — the region where raw is most wrong', () => {
|
|
for (const p of [0.80, 0.85, 0.90, 0.95]) {
|
|
expect(pc.inCertifiedRawBand(pc.MLB_HITS.certified_bands, p)).toBe(false);
|
|
}
|
|
});
|
|
|
|
it('holds ONE contract — no other sport or stat is certified', () => {
|
|
expect(Object.keys(pc.CONTRACTS)).toEqual(['mlb:hits']);
|
|
});
|
|
|
|
it('the held-out interval it records excludes zero', () => {
|
|
expect(pc.MLB_HITS.evidence.ci95[1]).toBeLessThan(0);
|
|
});
|
|
});
|
|
|
|
describe('the registry already asked the right question', () => {
|
|
it('serves() tests certified bands against the RAW p_win, not the output', () => {
|
|
const r = reg.createRegistry();
|
|
r.deploy('hits', { lodo_pass: true, ci: [-0.006, -0.001], map: { x: [0], y: [0] },
|
|
certified_bands: [[0.50, 0.80]] });
|
|
expect(r.serves('hits', 0.65).serve).toBe(true);
|
|
expect(r.serves('hits', 0.90).serve).toBe(false);
|
|
});
|
|
|
|
it('a refusal names the reason and never proposes raw', () => {
|
|
const r = reg.createRegistry();
|
|
r.deploy('hits', { lodo_pass: true, ci: [-0.006, -0.001], map: {}, certified_bands: [[0.50, 0.80]] });
|
|
const out = r.serves('hits', 0.92);
|
|
expect(out.serve).toBe(false);
|
|
expect(JSON.stringify(out)).not.toContain('0.92');
|
|
});
|
|
});
|
|
|
|
describe('probabilityContractService — loads, never fits', () => {
|
|
const registry = require('../../src/services/model/artifactRegistry');
|
|
|
|
it('returns the PROMOTED frozen artifact, not a fresh fit', async () => {
|
|
const b = await svc.build(null, { sport: 'mlb', stat: 'hits' });
|
|
expect(b).not.toBeNull();
|
|
expect(b.artifact.artifact_id).toBe(registry.PROMOTED['mlb:hits'].artifact_id);
|
|
expect(b.artifact.servable).toBe(true);
|
|
});
|
|
|
|
it('returns null — never a fallback curve — when nothing is promoted', async () => {
|
|
for (const stat of ['rbi', 'total_bases', 'runs']) {
|
|
expect(await svc.build(null, { sport: 'mlb', stat })).toBeNull();
|
|
}
|
|
expect(await svc.build(null, { sport: 'wnba', stat: 'points' })).toBeNull();
|
|
});
|
|
|
|
it('needs no database client at all — there is nothing left to read', async () => {
|
|
const b = await svc.build(undefined, { sport: 'mlb', stat: 'hits' });
|
|
expect(b.artifact.artifact_id).toBeTruthy();
|
|
});
|
|
|
|
it('two resolutions of the same input are identical, forever', async () => {
|
|
const b1 = await svc.build(null, { sport: 'mlb', stat: 'hits' });
|
|
const b2 = await svc.build(null, { sport: 'mlb', stat: 'hits' });
|
|
for (const p of [0.50, 0.55, 0.601, 0.72, 0.799]) {
|
|
const a = b1.resolve({ model_version: ERA, p_win: p });
|
|
const c = b2.resolve({ model_version: ERA, p_win: p });
|
|
expect(a.served_probability).toBe(c.served_probability);
|
|
expect(a.artifact.knot_digest).toBe(c.artifact.knot_digest);
|
|
}
|
|
});
|
|
});
|
|
|
|
describe('artifact identity — the frozen curve that actually ran', () => {
|
|
const registry = require('../../src/services/model/artifactRegistry');
|
|
const a = registry.load('mlb', 'hits');
|
|
|
|
it('carries every field needed to find and check it again', () => {
|
|
for (const k of ['artifact_id', 'procedure_version', 'sport', 'stat', 'model_version',
|
|
'fit_as_of', 'training_cutoff', 'fit_n', 'source_digest', 'algorithm', 'algorithm_version',
|
|
'knot_count', 'knot_digest', 'served_curve', 'served_curve_digest', 'certified_bands']) {
|
|
expect(a[k]).toBeDefined();
|
|
expect(a[k]).not.toBeNull();
|
|
}
|
|
});
|
|
|
|
it('fit_as_of and training_cutoff are DIFFERENT questions', () => {
|
|
// the eligibility bound, and the newest observation actually admitted
|
|
expect(a.fit_as_of).not.toBe(a.training_cutoff);
|
|
expect(a.training_cutoff < a.fit_as_of).toBe(true);
|
|
});
|
|
|
|
it('was fitted on the current era ONLY', () => {
|
|
expect(a.model_version).toBe(ERA);
|
|
expect(a.era_audit.wrong_era_rows).toBe(0);
|
|
expect(Object.keys(a.era_audit.era_counts)).toEqual([ERA]);
|
|
});
|
|
|
|
it('withheld nothing from the promoted fit', () => {
|
|
expect(a.withheld_from_fit).toBe(0);
|
|
expect(a.fit_n).toBe(a.era_audit.current_era_rows);
|
|
});
|
|
|
|
it('the served curve IS the served function over support, not a sample', () => {
|
|
for (let x = 0.50; x < 0.80 - 1e-9; x += 0.001) {
|
|
const raw = Math.round(x * 1000) / 1000;
|
|
let want = null;
|
|
for (const [from, val] of a.served_curve) { if (raw >= from) want = val; else break; }
|
|
expect(registry.applyCurve(a, raw)).toBe(want);
|
|
}
|
|
});
|
|
|
|
it('the curve covers ONLY certified support', () => {
|
|
for (const [from] of a.served_curve) {
|
|
expect(from).toBeGreaterThanOrEqual(0.50);
|
|
expect(from).toBeLessThan(0.80);
|
|
}
|
|
for (const p of [0.499, 0.80, 0.9, 0.99]) expect(registry.applyCurve(a, p)).toBeNull();
|
|
});
|
|
});
|
|
|
|
describe('point in time — a Read can only see settlements before its own day', () => {
|
|
const calSvc = require('../../src/services/model/calibrationService');
|
|
|
|
/** Records the filters actually applied, so this tests behaviour not source. */
|
|
function recordingClient(rows) {
|
|
const applied = [];
|
|
const q = {
|
|
select: () => q, eq: (c, v) => { applied.push(['eq', c, v]); return q; },
|
|
is: (c, v) => { applied.push(['is', c, v]); return q; },
|
|
in: (c, v) => { applied.push(['in', c, v]); return q; },
|
|
not: (c, o, v) => { applied.push(['not', c, o, v]); return q; },
|
|
lt: (c, v) => { applied.push(['lt', c, v]); return q; },
|
|
lte: (c, v) => { applied.push(['lte', c, v]); return q; },
|
|
gte: (c, v) => { applied.push(['gte', c, v]); return q; },
|
|
order: () => q, limit: () => q,
|
|
range: async () => ({ data: rows, error: null, count: rows.length }),
|
|
then: (res) => res({ data: rows, error: null }),
|
|
};
|
|
return { applied, client: { from: () => q } };
|
|
}
|
|
|
|
it('bounds the training walk STRICTLY BEFORE the cutoff, never at or after it', async () => {
|
|
const { applied, client } = recordingClient([]);
|
|
await calSvc.loadSettledRows(client, { sport: 'mlb', stat: 'hits', before: '2026-09-02' });
|
|
const bound = applied.filter((a) => a[1] === 'game_date');
|
|
expect(bound.length).toBeGreaterThan(0);
|
|
// strictly-less is the whole discipline: `lte` would admit same-day games
|
|
expect(bound.some((a) => a[0] === 'lt' && a[2] === '2026-09-02')).toBe(true);
|
|
expect(bound.some((a) => a[0] === 'lte')).toBe(false);
|
|
expect(bound.some((a) => a[0] === 'gte')).toBe(false);
|
|
});
|
|
|
|
it('the artifact names its own evidence horizon', async () => {
|
|
const built = await svc.build(null, { sport: 'mlb', stat: 'hits' });
|
|
expect(built.artifact.fit_as_of).toBeTruthy();
|
|
expect(built.artifact.training_cutoff).toBeTruthy();
|
|
// the bound, and the last date inside it — never the same claim
|
|
expect(built.artifact.training_cutoff < built.artifact.fit_as_of).toBe(true);
|
|
});
|
|
});
|