'use strict'; /** * A MONITORING CONTRACT WITH NO CALLSITE IS A PLAN. * * The previous tranche declared forward evaluation and shipped none. These * tests hold both halves: the evaluator behaves, AND production actually calls it. */ const fm = require('../../src/services/model/forwardMonitor'); const registry = require('../../src/services/model/artifactRegistry'); const A = registry.load('mlb', 'hits'); // Strictly after training_cutoff 2026-09-01. THREE dates, because the monitor // requires one certification-equivalent block of date evidence — a single-date // fixture can never reach a verdict, by design. const AFTER_DATES = ['2026-09-02', '2026-09-03', '2026-09-04']; const AFTER = AFTER_DATES[0]; /** Rows whose outcomes track the frozen curve, optionally shifted to force drift. */ function rows(n, shift = 0, over = {}) { return Array.from({ length: n }, (_, i) => { const raw = Math.round((0.50 + (i % 29) / 100) * 1000) / 1000; const served = registry.applyCurve(A, raw) ?? 0.6; return { p: raw, won: ((i * 2654435761) % 1000) / 1000 < served + shift ? 1 : 0, date: AFTER_DATES[i % AFTER_DATES.length], model_version: A.model_version, ...over }; }); } describe('the monitor evaluates and refuses to guess', () => { it('says HEALTHY only on enough evidence', () => { const r = fm.evaluate(A, rows(1200), registry.applyCurve); expect(r.health).toBe(fm.HEALTH.HEALTHY); expect(r.healthy).toBe(true); expect(r.n).toBeGreaterThanOrEqual(fm.MIN_FORWARD_ROWS); }); it('LOW N IS NOT HEALTHY AND NOT DRIFT — it is its own answer', () => { const r = fm.evaluate(A, rows(50), registry.applyCurve); expect(r.health).toBe(fm.HEALTH.INSUFFICIENT_SAMPLE); expect(r.healthy).toBeNull(); // never false, never true expect(r.reason).toContain(String(fm.MIN_FORWARD_ROWS)); }); it('the sample floor is DERIVED from the certified tolerance, not chosen', () => { // SE = sqrt(0.25/n) <= TOLERANCE/2 => n >= 0.25 / (TOLERANCE/2)^2 expect(fm.MIN_FORWARD_ROWS).toBe(Math.ceil(0.25 / ((fm.TOLERANCE / 2) ** 2))); expect(fm.TOLERANCE).toBe(0.05); // same number certifyBands used }); it('flags drift when the frozen curve stops matching outcomes', () => { const r = fm.evaluate(A, rows(1200, 0.20), registry.applyCurve); expect(r.health).toBe(fm.HEALTH.DRIFT_WARNING); expect(r.healthy).toBe(false); expect(r.drift_bands.length).toBeGreaterThan(0); }); it('refuses an unservable artifact and wrong-era evidence', () => { expect(fm.evaluate({ ...A, servable: false }, rows(1200), registry.applyCurve).health) .toBe(fm.HEALTH.INVALID); expect(fm.evaluate(A, rows(1200, 0, { model_version: 'engine1@2026-07-20' }), registry.applyCurve).health) .toBe(fm.HEALTH.INVALID); }); it('an unreadable read is AUDIT_UNAVAILABLE, never a health verdict', () => { const r = fm.evaluate(A, null, registry.applyCurve); expect(r.health).toBe(fm.HEALTH.AUDIT_UNAVAILABLE); expect(r.healthy).toBeNull(); expect(fm.evaluate(null, rows(1200), registry.applyCurve).health).toBe(fm.HEALTH.AUDIT_UNAVAILABLE); }); it('scores ONLY evidence the fit never saw', () => { const atCutoff = rows(1200).map((r) => ({ ...r, date: A.training_cutoff })); expect(fm.evaluate(A, atCutoff, registry.applyCurve).n).toBe(0); const before = rows(1200).map((r) => ({ ...r, date: '2026-08-01' })); expect(fm.evaluate(A, before, registry.applyCurve).n).toBe(0); }); it('scores ONLY rows inside certified support', () => { const outside = rows(1200).map((r, i) => ({ ...r, p: i % 2 ? 0.9 : 0.4 })); expect(fm.evaluate(A, outside, registry.applyCurve).n).toBe(0); }); }); describe('the monitor cannot change what it watches', () => { it('imports no fitter and performs no write', () => { const src = require('fs').readFileSync( require('path').join(__dirname, '../../src/services/model/forwardMonitor.js'), 'utf8'); const code = src.replace(/\/\*[\s\S]*?\*\//g, '').replace(/^\s*\/\/.*$/gm, ''); // token-level, not substring: 'promote' also appears inside the reason // string 'no promoted artifact', which is a description, not a call. for (const forbidden of ['fitIsotonic', 'fitPlatt', 'writeFileSync', '.upsert(', '.update(', 'promote(', 'artifactRegistry', 'PROMOTED']) { expect(code).not.toContain(forbidden); } }); it('evaluating does not mutate the artifact', () => { const before = JSON.stringify(A); fm.evaluate(A, rows(1200, 0.2), registry.applyCurve); expect(JSON.stringify(registry.load('mlb', 'hits'))).toBe(before); }); }); describe('alarm discipline', () => { it('does not alert on HEALTHY or INSUFFICIENT_SAMPLE', () => { for (const h of [fm.HEALTH.HEALTHY, fm.HEALTH.INSUFFICIENT_SAMPLE]) { expect(fm.monitorAlarm(null, { health: h, artifact_id: 'x' }).alert).toBe(false); } }); it('alerts once per state transition, not once per tick', () => { const first = fm.monitorAlarm(null, { health: fm.HEALTH.DRIFT_WARNING, artifact_id: 'x', reason: 'r' }); expect(first.alert).toBe(true); expect(fm.monitorAlarm(first.key, { health: fm.HEALTH.DRIFT_WARNING, artifact_id: 'x', reason: 'r' }).alert).toBe(false); }); it('says plainly that a drift alert has NOT changed the artifact', () => { const a = fm.monitorAlarm(null, { health: fm.HEALTH.DRIFT_WARNING, artifact_id: 'x', reason: 'r' }); expect(a.message).toMatch(/frozen and unchanged/); }); }); describe('THE CALLSITE — production actually runs it', () => { it('the scheduler exposes and invokes the monitor on its tick', async () => { const src = require('fs').readFileSync( require('path').join(__dirname, '../../src/snapshotScheduler.js'), 'utf8'); expect(src).toContain('await calibrationMonitorTick();'); expect(src).toContain('calibrationMonitorTick };'); }); it('the tick calls evaluate with the promoted artifact and forward-only rows', async () => { const sched = require('../../src/snapshotScheduler'); const calls = []; const fake = { monitorDue: () => true, evaluate: (artifact, rws) => { calls.push({ artifact, rws }); return { health: 'HEALTHY', healthy: true, n: 999, min_required: 400, artifact_id: artifact.artifact_id }; }, monitorAlarm: () => ({ alert: false, key: 'k' }), HEALTH: fm.HEALTH, }; const loaded = []; const prev = process.env.SNAPSHOT_CRON; process.env.SNAPSHOT_CRON = '1'; // the scheduler is inert unless armed const s = sched.startSnapshotScheduler({ forwardMonitor: fake, artifactRegistry: registry, currentEraSource: { loadRows: async (sb, args) => { loaded.push(args); return rows(500); } }, supabase: {}, now: () => new Date('2026-09-03T05:00:00Z'), runAllSnapshots: async () => ({}), }); try { expect(s).not.toBeNull(); // armed, or the assertions below are vacuous await s.calibrationMonitorTick(); expect(calls).toHaveLength(1); expect(calls[0].artifact.artifact_id).toBe(A.artifact_id); // forward-only: bounded strictly after the artifact's training cutoff expect(loaded[0].after).toBe(A.training_cutoff); expect(loaded[0].modelVersion).toBe(A.model_version); } finally { if (s && s.interval) clearInterval(s.interval); if (prev === undefined) delete process.env.SNAPSHOT_CRON; else process.env.SNAPSHOT_CRON = prev; } }); }); describe('row count must not masquerade as temporal evidence', () => { const fp = require('../../src/services/model/fitPolicy'); const many = (n, dates) => Array.from({ length: n }, (_, i) => { const raw = Math.round((0.50 + (i % 29) / 100) * 1000) / 1000; const served = registry.applyCurve(A, raw) ?? 0.6; return { p: raw, won: ((i * 2654435761) % 1000) / 1000 < served ? 1 : 0, date: dates[i % dates.length], model_version: A.model_version }; }); it('the date floor is READ FROM the procedure, not invented here', () => { expect(fm.MIN_FORWARD_DATES).toBe(fp.POLICY_V1.certification.eval_block_dates); expect(fm.MIN_FORWARD_DATES).toBe(3); // the walk-forward's fold width }); it('1,200 rows from ONE slate is not a verdict', () => { const r = fm.evaluate(A, many(1200, ['2026-09-02']), registry.applyCurve); expect(r.n).toBeGreaterThanOrEqual(fm.MIN_FORWARD_ROWS); // rows are ample expect(r.health).toBe(fm.HEALTH.INSUFFICIENT_SAMPLE); // dates are not expect(r.healthy).toBeNull(); expect(r.settled_date_count).toBe(1); expect(r.reason).toContain('settled date'); }); it('two dates is still not enough', () => { const r = fm.evaluate(A, many(1200, ['2026-09-02', '2026-09-03']), registry.applyCurve); expect(r.health).toBe(fm.HEALTH.INSUFFICIENT_SAMPLE); expect(r.settled_date_count).toBe(2); }); it('one certification-equivalent block of dates unlocks a verdict', () => { const r = fm.evaluate(A, many(1200, ['2026-09-02', '2026-09-03', '2026-09-04']), registry.applyCurve); expect(r.settled_date_count).toBe(3); expect([fm.HEALTH.HEALTHY, fm.HEALTH.DRIFT_WARNING]).toContain(r.health); }); it('enough dates does NOT rescue a thin row count', () => { const r = fm.evaluate(A, many(50, ['2026-09-02', '2026-09-03', '2026-09-04']), registry.applyCurve); expect(r.health).toBe(fm.HEALTH.INSUFFICIENT_SAMPLE); expect(r.healthy).toBeNull(); }); it('the date count is computed from the SCORED rows, not from undefined', () => { // Without `date` on the scored row every row hashes to undefined and the // count is always 1 — the gate would look right while measuring nothing. const r = fm.evaluate(A, many(900, ['2026-09-02', '2026-09-03', '2026-09-04']), registry.applyCurve); expect(r.settled_dates).toEqual(['2026-09-02', '2026-09-03', '2026-09-04']); }); });