Files
vyndr/tests/unit/calibrationRegistry.test.js
T
builtbykev ced40421ed Audit the LODO instrument: it cannot evaluate any stat, and both prior
FAILs were false

PHASE 0 — the gate at 1f40014 was mine and was an incoherent pair. A 1-SE
informativeness bar with a ZERO-reversal rule: at exactly 1 SE a stable
stat's drop reverses with prob Phi(-1)=0.1587, so on four informative
drops P(>=1 reversal | perfectly stable) = 1 - 0.8413^4 = 0.50. It failed
stable stats half the time by construction. And the pooled n*=70
mis-credited EVERY stat -- too low for hits (own 77) and runs (81), too
high for total_bases (60) and rbi (54).

PHASE 1, blind. Per-stat (g, sigma_row): hits -0.01288/0.11251, TB
-0.01380/0.10680, rbi -0.00884/0.06459, runs -0.00902/0.08080. All four
clear z=1.96 at full n, so none is NO-EFFECT. Committed k=1 with per-stat
n* and a binomial cutoff holding FP at 0.004-0.031.

THE FINDING THAT DOMINATES: the test has no power. Against a strong
instability (date-to-date SD equal to the effect) it detects a failure
1.4%-9.3% of the time, and across every k from 1.0 to 2.0 the best any
stat reaches is 0.337. A gate that cannot fail cannot pass, so
LODO_POWER_FLOOR=0.50 makes UNTESTABLE structural -- "could not test" can
never read as "passed".

PHASE 2/3 cold, at each stat's OWN n*:

  hits  5 informative, 0 reversals, cutoff 2, power 0.093  UNTESTABLE
  TB    5 informative, 0 reversals, cutoff 2, power 0.093  UNTESTABLE
  rbi   4 informative, 1 reversal,  cutoff 2, power 0.045  UNTESTABLE
  runs  3 informative, 2 reversals, cutoff 2, power 0.014  UNTESTABLE

Setting the power floor aside entirely, NOT ONE STAT EXCEEDS ITS CUTOFF.

PHASE 4 — rbi's FAIL was false, as the order suspected. So was RUNS' --
which the order did not anticipate, having classified it DATE-DRIVEN on a
244-row reversal; two reversals in three drops does not clear a cutoff of
2. TB's PASS was vacuous: the test could not have failed it. hits' own n*
is LARGER than the pooled one (77 vs 70), and it remains untestable.

PHASE 5 — deploy basis is now the date-clustered CI alone:

  hits  CI [-0.0139,-0.0097], 4 date clusters   relabelled ci_only
  TB    CI [-0.0061,-0.0045], 2 date clusters   RELABELLED, kept
  rbi   CI [-0.0092,-0.0010], 2 date clusters   NEWLY DEPLOYED
  runs  no fittable map at its split            REFUSE, no CI either

Every deployed stat carries calibration_basis ci_only_lodo_untestable and
auto-demotion is the SOLE stability guard, not a backstop to a passed
test. Stated plainly: those intervals rest on 2-4 date clusters, which is
thin, and it is now the only support. rbi gains chainAcross stackability;
its bands rebuilt on p_win_calibrated (425 rows) are every-archetype
base_rate. runs is queued for the low-param calibrator for the ordinary
reason -- no fittable map -- not on the date-driven finding, which was an
artefact.

PHASE 6 — the deploy set was set by a coin-flip-power ruler; it is now set
by a per-stat power-coherent pre-committed test whose first act was to
report that it cannot evaluate anything. The audit was permitted to wound
the live deploy and did: total_bases lost its LODO claim. Standing
question unchanged -- 18 archetype slots across three deployed stats, every
one a single band indistinguishable from base rate.

Blind ordering held. p_win never mutated. No Bonferroni slot. Counter and
frozen clusters verified file-by-file.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01W1sivYNqY2TS5ftykmHBU9
2026-08-06 23:01:57 -04:00

177 lines
7.2 KiB
JavaScript

'use strict';
/**
* Which stats may serve a calibrated number.
*
* The thing these protect is the meaning of PROVISIONAL: a provisional deploy
* that cannot be taken away is just a deploy.
*/
const { createRegistry, STATUS, PROMOTION_DATE_CLUSTERS } = require('../../src/services/model/calibrationRegistry');
const MAP = [{ lo: 0.5, hi: 0.7, value: 0.55, n: 300 }];
const GOOD = { lodo_pass: true, ci: [-0.0061, -0.0045], map: MAP, certified_bands: [[0.6, 0.8]], date_clusters: 7, at: '2026-08-06' };
describe('deploy needs BOTH gates', () => {
it('deploys when LODO passes and the interval excludes zero', () => {
const r = createRegistry();
expect(r.deploy('total_bases', GOOD).status).toBe(STATUS.PROVISIONAL);
});
it('refuses on a LODO failure however good the interval', () => {
const r = createRegistry();
const out = r.deploy('runs', { ...GOOD, lodo_pass: false });
expect(out.ok).toBe(false);
expect(out.reason).toMatch(/date-driven/);
});
it('refuses when the interval spans zero however clean the LODO', () => {
const r = createRegistry();
const out = r.deploy('hits', { ...GOOD, ci: [-0.01, 0.002] });
expect(out.ok).toBe(false);
expect(out.reason).toMatch(/does not exclude zero/);
});
it('refuses without a map — there is nothing to serve', () => {
const r = createRegistry();
expect(r.deploy('hits', { ...GOOD, map: null }).ok).toBe(false);
});
});
describe('auto-demotion is what makes provisional honest', () => {
it('demotes on the first date where the interval stops excluding zero', () => {
const r = createRegistry();
r.deploy('total_bases', GOOD);
const out = r.reverify('total_bases', { ci: [-0.004, 0.001], date: '2026-08-07' });
expect(out.status).toBe(STATUS.NONE);
expect(out.reason).toBe('ci_no_longer_excludes_zero');
expect(out.breaking_date).toBe('2026-08-07');
expect(r.serves('total_bases', 0.65).serve).toBe(false);
});
it('demotes when the favourite over-prediction flips sign', () => {
// A flip means the correction is now pushing the wrong way.
const r = createRegistry();
r.deploy('total_bases', GOOD);
const out = r.reverify('total_bases', { ci: [-0.006, -0.004], favourite_bias: -0.03, date: '2026-08-08' });
expect(out.status).toBe(STATUS.NONE);
expect(out.reason).toBe('favourite_bias_flipped');
});
it('logs the demotion with its breaking date', () => {
const r = createRegistry();
r.deploy('total_bases', GOOD);
r.reverify('total_bases', { ci: [0.001, 0.004], date: '2026-08-09' });
const ev = r.log().find((e) => e.event === 'auto_demoted');
expect(ev).toMatchObject({ stat: 'total_bases', at: '2026-08-09' });
});
it('stays deployed while both conditions hold', () => {
const r = createRegistry();
r.deploy('total_bases', GOOD);
const out = r.reverify('total_bases', { ci: [-0.007, -0.003], favourite_bias: 0.17, date: '2026-08-07' });
expect(out.status).toBe(STATUS.PROVISIONAL);
expect(out.changed).toBe(false);
});
});
describe('the >=40 date-cluster bar is the PROMOTION bar, not the deploy bar', () => {
it('does not block deployment', () => {
const r = createRegistry();
expect(r.deploy('total_bases', { ...GOOD, date_clusters: 7 }).ok).toBe(true);
});
it('promotes out of provisional once it is met', () => {
const r = createRegistry();
r.deploy('total_bases', GOOD);
const out = r.reverify('total_bases', { ci: [-0.006, -0.004], date_clusters: PROMOTION_DATE_CLUSTERS, date: '2026-09-15' });
expect(out.status).toBe(STATUS.PROMOTED);
});
it('does not promote while the interval has stopped holding', () => {
const r = createRegistry();
r.deploy('total_bases', GOOD);
const out = r.reverify('total_bases', { ci: [-0.001, 0.003], date_clusters: 60, date: '2026-09-15' });
expect(out.status).toBe(STATUS.NONE);
});
});
describe('serving is band-limited', () => {
it('serves inside the certified band and refuses outside it', () => {
const r = createRegistry();
r.deploy('total_bases', GOOD);
expect(r.serves('total_bases', 0.65).serve).toBe(true);
expect(r.serves('total_bases', 0.65).provisional).toBe(true);
expect(r.serves('total_bases', 0.95).serve).toBe(false);
expect(r.serves('total_bases', 0.95).reason).toMatch(/outside the certified band/);
});
it('an undeployed stat never serves', () => {
const r = createRegistry();
expect(r.serves('hits', 0.6).serve).toBe(false);
expect(r.serves('hits', 0.6).reason).toBe('not deployed');
});
it('a missing p_win serves nothing', () => {
const r = createRegistry();
r.deploy('total_bases', GOOD);
expect(r.serves('total_bases', null).serve).toBe(false);
});
});
describe('the LODO test is a COHERENT pair, not a bar plus an unrelated rule', () => {
const { LODO_K, LODO_TEST, LODO_POWER_FLOOR } = require('../../src/services/model/calibrationRegistry');
const normCdf = (z) => {
const t = 1 / (1 + 0.2316419 * Math.abs(z));
const d = 0.3989422804014327 * Math.exp(-z * z / 2);
const p = d * t * (0.319381530 + t * (-0.356563782 + t * (1.781477937 + t * (-1.821255978 + t * 1.330274429))));
return z >= 0 ? 1 - p : p;
};
const binomPmf = (n, k, p) => {
let logC = 0;
for (let i = 0; i < k; i += 1) logC += Math.log(n - i) - Math.log(i + 1);
return Math.exp(logC + k * Math.log(p) + (n - k) * Math.log(1 - p));
};
const tail = (n, c, p) => { let s = 0; for (let k = c + 1; k <= n; k += 1) s += binomPmf(n, k, p); return s; };
it('each n* is what its own (sigma_row, g) produce — no pooled value', () => {
for (const [stat, t] of Object.entries(LODO_TEST)) {
expect(Math.ceil(LODO_K ** 2 * (t.sigma_row / Math.abs(t.g)) ** 2)).toBe(t.n_star);
}
// And the four differ, which is exactly why one pooled number mis-credited them.
const stars = Object.values(LODO_TEST).map((t) => t.n_star);
expect(new Set(stars).size).toBeGreaterThan(1);
});
it('each cutoff is the smallest one holding the false-positive rate at 0.05', () => {
const p = normCdf(-LODO_K);
for (const [stat, t] of Object.entries(LODO_TEST)) {
expect(tail(t.informative_drops, t.cutoff, p)).toBeLessThanOrEqual(0.05);
if (t.cutoff > 0) expect(tail(t.informative_drops, t.cutoff - 1, p)).toBeGreaterThan(0.05);
expect(t.fp).toBeCloseTo(tail(t.informative_drops, t.cutoff, p), 3);
}
});
it('the OLD rule is demonstrably incoherent — it failed stable stats ~half the time', () => {
// Zero-reversal rule at a 1-SE bar, on four informative drops.
const pNoise = normCdf(-1);
const falseFail = 1 - (1 - pNoise) ** 4;
expect(falseFail).toBeGreaterThan(0.45);
expect(falseFail).toBeLessThan(0.55);
});
it('every stat falls below the power floor, so none may claim LODO stability', () => {
// A gate that cannot fail is not a gate. This is the honest state at this
// date count, and the floor makes it structural rather than a footnote.
for (const [stat, t] of Object.entries(LODO_TEST)) {
expect(t.power).toBeLessThan(LODO_POWER_FLOOR);
}
});
it('the stale pooled threshold is nulled so nothing can read it', () => {
const { LODO_MIN_HELD_ROWS } = require('../../src/services/model/calibrationRegistry');
expect(LODO_MIN_HELD_ROWS).toBeNull();
});
});