The monitoring contract gets a callsite

Traced 2026-09-03: `calibrationRegistry.reverify` had ZERO production callers,
and the only reference to the artifact machinery outside its own directory was
the shadow builder, behind a flag that is off. The previous tranche declared
that future settled outcomes are forward evaluation evidence and shipped no
evaluator — the contract lived in a comment. That is the same shape as
statModel.js and correlateValidator.js, cited for months and never present.

`forwardMonitor` scores the FROZEN artifact on outcomes settled strictly after
its training_cutoff, and `snapshotScheduler.calibrationMonitorTick` runs it on
the existing per-minute cadence, throttled 6h, every failure swallowed. A test
asserts the tick is defined, invoked AND exported, and drives it with a fake to
prove it passes the promoted artifact and a forward-only window.

It cannot change what it watches: the module imports no fitter and no registry,
and a test greps the stripped source for fitIsotonic, fitPlatt, writeFileSync,
upsert, update, promote( , artifactRegistry and PROMOTED.

LOW N IS ITS OWN ANSWER. Below the floor it reports INSUFFICIENT_SAMPLE with
`healthy: null` — never false, never true. The floor is DERIVED, not chosen:
resolving an error of the certified tolerance at two standard errors needs
n >= 0.25/(0.05/2)^2 = 400, and a test recomputes it from TOLERANCE so the two
cannot drift apart. TOLERANCE is 0.05, the same number certifyBands used, so the
monitor can be neither stricter nor laxer than the thing it watches.

Wrong-era forward rows are INVALID, not scored — scoring them would measure a
different forecaster, which is the defect this line of work removed. An
unreadable read is AUDIT_UNAVAILABLE, which is not a health verdict. A drift
alert says in its own text that the artifact is frozen and unchanged, because
the alert is not a demotion.

`currentEraSource.loadRows` gains an optional `after` bound, strictly greater
so the cutoff date itself can never be scored as forward evidence.

SHADOW REMAINS BLOCKED: the production variable is still absent
(configuration_source "default") after a restart at 05:09:02Z, so no shadow
cohort was obtained and no replay was substituted for one.

Artifact unchanged: mlb-hits-isotonic@2026-09-03, knots 5ae940ea163b7da2.
Live OFF. Suite 405/405, 5,654 passed. Teeth 31/31 + 10/10 + 23/23.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CQJeAG8vcDoL5zkiaJyVb8
This commit is contained in:
Kev
2026-09-03 01:29:47 -04:00
parent f4f51ed7ca
commit 044df7a426
6 changed files with 477 additions and 4 deletions
+160
View File
@@ -0,0 +1,160 @@
'use strict';
/**
* A MONITORING CONTRACT WITH NO CALLSITE IS A PLAN.
*
* The previous tranche declared forward evaluation and shipped none. These
* tests hold both halves: the evaluator behaves, AND production actually calls it.
*/
const fm = require('../../src/services/model/forwardMonitor');
const registry = require('../../src/services/model/artifactRegistry');
const A = registry.load('mlb', 'hits');
const AFTER = '2026-09-02'; // strictly after training_cutoff 2026-09-01
/** Rows whose outcomes track the frozen curve, optionally shifted to force drift. */
function rows(n, shift = 0, over = {}) {
return Array.from({ length: n }, (_, i) => {
const raw = Math.round((0.50 + (i % 29) / 100) * 1000) / 1000;
const served = registry.applyCurve(A, raw) ?? 0.6;
return { p: raw, won: ((i * 2654435761) % 1000) / 1000 < served + shift ? 1 : 0,
date: AFTER, model_version: A.model_version, ...over };
});
}
describe('the monitor evaluates and refuses to guess', () => {
it('says HEALTHY only on enough evidence', () => {
const r = fm.evaluate(A, rows(1200), registry.applyCurve);
expect(r.health).toBe(fm.HEALTH.HEALTHY);
expect(r.healthy).toBe(true);
expect(r.n).toBeGreaterThanOrEqual(fm.MIN_FORWARD_ROWS);
});
it('LOW N IS NOT HEALTHY AND NOT DRIFT — it is its own answer', () => {
const r = fm.evaluate(A, rows(50), registry.applyCurve);
expect(r.health).toBe(fm.HEALTH.INSUFFICIENT_SAMPLE);
expect(r.healthy).toBeNull(); // never false, never true
expect(r.reason).toContain(String(fm.MIN_FORWARD_ROWS));
});
it('the sample floor is DERIVED from the certified tolerance, not chosen', () => {
// SE = sqrt(0.25/n) <= TOLERANCE/2 => n >= 0.25 / (TOLERANCE/2)^2
expect(fm.MIN_FORWARD_ROWS).toBe(Math.ceil(0.25 / ((fm.TOLERANCE / 2) ** 2)));
expect(fm.TOLERANCE).toBe(0.05); // same number certifyBands used
});
it('flags drift when the frozen curve stops matching outcomes', () => {
const r = fm.evaluate(A, rows(1200, 0.20), registry.applyCurve);
expect(r.health).toBe(fm.HEALTH.DRIFT_WARNING);
expect(r.healthy).toBe(false);
expect(r.drift_bands.length).toBeGreaterThan(0);
});
it('refuses an unservable artifact and wrong-era evidence', () => {
expect(fm.evaluate({ ...A, servable: false }, rows(1200), registry.applyCurve).health)
.toBe(fm.HEALTH.INVALID);
expect(fm.evaluate(A, rows(1200, 0, { model_version: 'engine1@2026-07-20' }), registry.applyCurve).health)
.toBe(fm.HEALTH.INVALID);
});
it('an unreadable read is AUDIT_UNAVAILABLE, never a health verdict', () => {
const r = fm.evaluate(A, null, registry.applyCurve);
expect(r.health).toBe(fm.HEALTH.AUDIT_UNAVAILABLE);
expect(r.healthy).toBeNull();
expect(fm.evaluate(null, rows(1200), registry.applyCurve).health).toBe(fm.HEALTH.AUDIT_UNAVAILABLE);
});
it('scores ONLY evidence the fit never saw', () => {
const atCutoff = rows(1200).map((r) => ({ ...r, date: A.training_cutoff }));
expect(fm.evaluate(A, atCutoff, registry.applyCurve).n).toBe(0);
const before = rows(1200).map((r) => ({ ...r, date: '2026-08-01' }));
expect(fm.evaluate(A, before, registry.applyCurve).n).toBe(0);
});
it('scores ONLY rows inside certified support', () => {
const outside = rows(1200).map((r, i) => ({ ...r, p: i % 2 ? 0.9 : 0.4 }));
expect(fm.evaluate(A, outside, registry.applyCurve).n).toBe(0);
});
});
describe('the monitor cannot change what it watches', () => {
it('imports no fitter and performs no write', () => {
const src = require('fs').readFileSync(
require('path').join(__dirname, '../../src/services/model/forwardMonitor.js'), 'utf8');
const code = src.replace(/\/\*[\s\S]*?\*\//g, '').replace(/^\s*\/\/.*$/gm, '');
// token-level, not substring: 'promote' also appears inside the reason
// string 'no promoted artifact', which is a description, not a call.
for (const forbidden of ['fitIsotonic', 'fitPlatt', 'writeFileSync', '.upsert(', '.update(',
'promote(', 'artifactRegistry', 'PROMOTED']) {
expect(code).not.toContain(forbidden);
}
});
it('evaluating does not mutate the artifact', () => {
const before = JSON.stringify(A);
fm.evaluate(A, rows(1200, 0.2), registry.applyCurve);
expect(JSON.stringify(registry.load('mlb', 'hits'))).toBe(before);
});
});
describe('alarm discipline', () => {
it('does not alert on HEALTHY or INSUFFICIENT_SAMPLE', () => {
for (const h of [fm.HEALTH.HEALTHY, fm.HEALTH.INSUFFICIENT_SAMPLE]) {
expect(fm.monitorAlarm(null, { health: h, artifact_id: 'x' }).alert).toBe(false);
}
});
it('alerts once per state transition, not once per tick', () => {
const first = fm.monitorAlarm(null, { health: fm.HEALTH.DRIFT_WARNING, artifact_id: 'x', reason: 'r' });
expect(first.alert).toBe(true);
expect(fm.monitorAlarm(first.key, { health: fm.HEALTH.DRIFT_WARNING, artifact_id: 'x', reason: 'r' }).alert).toBe(false);
});
it('says plainly that a drift alert has NOT changed the artifact', () => {
const a = fm.monitorAlarm(null, { health: fm.HEALTH.DRIFT_WARNING, artifact_id: 'x', reason: 'r' });
expect(a.message).toMatch(/frozen and unchanged/);
});
});
describe('THE CALLSITE — production actually runs it', () => {
it('the scheduler exposes and invokes the monitor on its tick', async () => {
const src = require('fs').readFileSync(
require('path').join(__dirname, '../../src/snapshotScheduler.js'), 'utf8');
expect(src).toContain('await calibrationMonitorTick();');
expect(src).toContain('calibrationMonitorTick };');
});
it('the tick calls evaluate with the promoted artifact and forward-only rows', async () => {
const sched = require('../../src/snapshotScheduler');
const calls = [];
const fake = {
monitorDue: () => true,
evaluate: (artifact, rws) => { calls.push({ artifact, rws }); return { health: 'HEALTHY', healthy: true, n: 999, min_required: 400, artifact_id: artifact.artifact_id }; },
monitorAlarm: () => ({ alert: false, key: 'k' }),
HEALTH: fm.HEALTH,
};
const loaded = [];
const prev = process.env.SNAPSHOT_CRON;
process.env.SNAPSHOT_CRON = '1'; // the scheduler is inert unless armed
const s = sched.startSnapshotScheduler({
forwardMonitor: fake,
artifactRegistry: registry,
currentEraSource: { loadRows: async (sb, args) => { loaded.push(args); return rows(500); } },
supabase: {},
now: () => new Date('2026-09-03T05:00:00Z'),
runAllSnapshots: async () => ({}),
});
try {
expect(s).not.toBeNull(); // armed, or the assertions below are vacuous
await s.calibrationMonitorTick();
expect(calls).toHaveLength(1);
expect(calls[0].artifact.artifact_id).toBe(A.artifact_id);
// forward-only: bounded strictly after the artifact's training cutoff
expect(loaded[0].after).toBe(A.training_cutoff);
expect(loaded[0].modelVersion).toBe(A.model_version);
} finally {
if (s && s.interval) clearInterval(s.interval);
if (prev === undefined) delete process.env.SNAPSHOT_CRON; else process.env.SNAPSHOT_CRON = prev;
}
});
});