f61ec6b391
Seven orders of measurement-first repair. The served grade does not move. A0/A1 — the unordered page walk returned the right COUNT and the wrong ROWS: 410-617 of 2,490 duplicated with an equal number never returned, while rows.length matched the server exactly. safePaginate orders on a real unique key, verifies the tuple at runtime, and THROWS on a query error instead of treating it as end-of-data. Both hits PROVES are withdrawn: they were drawn through that reader, and defense_by_direction's distinct-n was likely below the gate floor all along. A2/A2b — rolled across every reader: 11 FAIL -> 0. Composite keys pulled from pg_index (the context tables are dated-composite and had no single unique column). The unordered helper is deleted, not parked. A3 — ledgerService and retentionService defaulted the SAME env var to DIFFERENT versions, so no ledger row ever carried the marker eligibility requires. One source now. model_snapshots settlement moved onto the cron: 15,484 -> 28,894 settled, repaired-champion 0 -> 7,556. A4 — hitsFactorContext takes an as-of cutoff. Refusal over reconstruction: no row at-or-before the date means the factor does not apply, never the nearest row. Live path unchanged, proven 400/400 on real rows. A5 — factor_inputs freezes what the factor READ, never the multiplier, so an audit can recompute and check. It also recorded the finding: the three hits factors have NEVER fired. prop.opponent and prop.opposing_pitcher are read by the resolver and written by nothing. A6/A7 — matchupKeys resolves those keys from the posted lineup plus the schedule's probable pitchers, and fires the factors into a SHADOW freeze: 248 fires on 308 props, 245 of which would move the grade. The served forecast is untouched. specs/a8-shadow-factor-gate.md pre-registers the test that decides whether they ever go live. Nothing is turned on. CALIBRATION_DEPLOYED stays []. Both verdicts stay withdrawn. 4,772 tests / 371 suites green, web build exit 0, read-integrity harness 34/34. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
204 lines
7.7 KiB
JavaScript
204 lines
7.7 KiB
JavaScript
'use strict';
|
|
|
|
/**
|
|
* Read-integrity harness — spec: specs/read-integrity-harness.md §5.
|
|
* Every case uses a FAKE page fetcher; nothing here touches a network.
|
|
*/
|
|
|
|
const ri = require('../../src/utils/readIntegrity');
|
|
|
|
/** A fetcher that serves a fixed list of pages, then empties. */
|
|
const pagesOf = (pages) => async (from, to) => {
|
|
const size = to - from + 1;
|
|
const idx = Math.floor(from / size);
|
|
return pages[idx] || [];
|
|
};
|
|
|
|
const row = (id) => ({ id });
|
|
const ids = (a, b) => Array.from({ length: b - a + 1 }, (_, i) => row(a + i));
|
|
|
|
describe('walk', () => {
|
|
it('stops on a short page', async () => {
|
|
const rows = await ri.walk(pagesOf([ids(1, 10), ids(11, 15)]), 10);
|
|
expect(rows).toHaveLength(15);
|
|
});
|
|
|
|
it('stops on an empty page after a full one', async () => {
|
|
const rows = await ri.walk(pagesOf([ids(1, 10), []]), 10);
|
|
expect(rows).toHaveLength(10);
|
|
});
|
|
|
|
it('keeps duplicates — collapsing them here would erase the measurement', async () => {
|
|
const rows = await ri.walk(pagesOf([ids(1, 10), [row(3), row(4)]]), 10);
|
|
expect(rows).toHaveLength(12);
|
|
});
|
|
|
|
it('PROPAGATES a fetch error instead of treating it as end-of-data', async () => {
|
|
// The calibrationService.fromLedger:117 defect: `if (error || !data) break`
|
|
// makes a partial read indistinguishable from a complete one.
|
|
const boom = async () => { throw new Error('fetch failed'); };
|
|
await expect(ri.walk(boom, 10)).rejects.toThrow('fetch failed');
|
|
});
|
|
|
|
it('refuses to report a runaway walk as complete', async () => {
|
|
const never = async () => ids(1, 10); // always a full page
|
|
await expect(ri.walk(never, 10, 3)).rejects.toThrow(/maxPages/);
|
|
});
|
|
});
|
|
|
|
describe('analyze — the verdict is measured, never inferred', () => {
|
|
it('PASSes a walk that is set-identical to the control', () => {
|
|
const r = ri.analyze({ exactCount: 15, unordered: ids(1, 15), ordered: ids(1, 15) });
|
|
expect(r.verdict).toBe(ri.VERDICT.PASS);
|
|
expect(r.unordered.duplicates).toBe(0);
|
|
expect(r.corruption_pct).toBe(0);
|
|
});
|
|
|
|
it('FAILs the measured production shape: last page re-emits earlier rows', () => {
|
|
// 2,490-row read reproduced in miniature: the tail page repeats rows already
|
|
// returned, so an equal number of real rows are never seen.
|
|
const unordered = [...ids(1, 10), ...ids(11, 20), ...[row(3), row(4), row(5)]];
|
|
const ordered = ids(1, 23);
|
|
const r = ri.analyze({ exactCount: 23, unordered, ordered });
|
|
expect(r.verdict).toBe(ri.VERDICT.FAIL);
|
|
expect(r.unordered.rows).toBe(23); // the count looks perfect
|
|
expect(r.unordered.distinct).toBe(20); // the rows are not
|
|
expect(r.unordered.duplicates).toBe(3);
|
|
expect(r.unordered.missing_vs_control).toBe(3);
|
|
expect(r.corruption_pct).toBeCloseTo(13.0, 1);
|
|
});
|
|
|
|
it('META-SCAR GUARD: an ORDERED walk that still duplicates is FAIL, not PASS', () => {
|
|
// The clause is not the result. A harness that credited the presence of an
|
|
// ORDER BY would be verifying its own intention.
|
|
const orderedButBroken = [...ids(1, 20), row(20)];
|
|
const r = ri.analyze({ exactCount: 21, unordered: orderedButBroken, ordered: ids(1, 21) });
|
|
expect(r.verdict).toBe(ri.VERDICT.FAIL);
|
|
expect(r.unordered.duplicates).toBe(1);
|
|
});
|
|
|
|
it('reports CONTROL_INVALID when the control is short of the server count', () => {
|
|
const r = ri.analyze({ exactCount: 30, unordered: ids(1, 25), ordered: ids(1, 25) });
|
|
expect(r.verdict).toBe(ri.VERDICT.CONTROL_INVALID);
|
|
expect(r.control_valid).toBe(false);
|
|
});
|
|
|
|
it('never issues PASS without a server count to validate against', () => {
|
|
const r = ri.analyze({ exactCount: null, unordered: ids(1, 5), ordered: ids(1, 5) });
|
|
expect(r.verdict).not.toBe(ri.VERDICT.PASS);
|
|
expect(r.verdict).toBe(ri.VERDICT.CONTROL_INVALID);
|
|
});
|
|
|
|
it('reports KEY_NOT_UNIQUE rather than blaming the walk for duplicate data', () => {
|
|
// batter_spray keyed on player_key|as_of_date has genuine duplicate keys.
|
|
const dup = [{ k: 'a' }, { k: 'a' }, { k: 'b' }];
|
|
const r = ri.analyze({ exactCount: 3, unordered: dup, ordered: dup, keyCols: ['k'] });
|
|
expect(r.verdict).toBe(ri.VERDICT.KEY_NOT_UNIQUE);
|
|
expect(r.key_unique).toBe(false);
|
|
});
|
|
|
|
it('counts a composite key across columns', () => {
|
|
const rows = [{ a: 1, b: 'x' }, { a: 1, b: 'y' }];
|
|
const r = ri.analyze({ exactCount: 2, unordered: rows, ordered: rows, keyCols: ['a', 'b'] });
|
|
expect(r.verdict).toBe(ri.VERDICT.PASS);
|
|
});
|
|
|
|
it('distinguishes an EXTRA row from a missing one', () => {
|
|
const r = ri.analyze({ exactCount: 5, unordered: [...ids(1, 5), row(99)], ordered: ids(1, 5) });
|
|
expect(r.unordered.extra_vs_control).toBe(1);
|
|
expect(r.unordered.missing_vs_control).toBe(0);
|
|
expect(r.verdict).toBe(ri.VERDICT.FAIL);
|
|
});
|
|
});
|
|
|
|
describe('applyFilters', () => {
|
|
const fake = () => {
|
|
const calls = [];
|
|
const q = {
|
|
calls,
|
|
eq: (...a) => { calls.push(['eq', ...a]); return q; },
|
|
is: (...a) => { calls.push(['is', ...a]); return q; },
|
|
in: (...a) => { calls.push(['in', ...a]); return q; },
|
|
not: (...a) => { calls.push(['not', ...a]); return q; },
|
|
lt: (...a) => { calls.push(['lt', ...a]); return q; },
|
|
gte: (...a) => { calls.push(['gte', ...a]); return q; },
|
|
};
|
|
return q;
|
|
};
|
|
|
|
it('chains ops in order', () => {
|
|
const q = ri.applyFilters(fake(), [
|
|
['eq', 'sport', 'mlb'],
|
|
['is', 'user_id', null],
|
|
['in', 'outcome', ['hit', 'miss']],
|
|
['not', 'p_win', 'is', null],
|
|
['lt', 'game_date', '2026-08-09'],
|
|
]);
|
|
expect(q.calls).toEqual([
|
|
['eq', 'sport', 'mlb'],
|
|
['is', 'user_id', null],
|
|
['in', 'outcome', ['hit', 'miss']],
|
|
['not', 'p_win', 'is', null],
|
|
['lt', 'game_date', '2026-08-09'],
|
|
]);
|
|
});
|
|
|
|
it('THROWS on an unknown op rather than skipping it — a dropped filter changes the row set', () => {
|
|
expect(() => ri.applyFilters(fake(), [['contains', 'x', 'y']])).toThrow(/unsupported filter op/);
|
|
});
|
|
|
|
it('handles an empty filter list', () => {
|
|
expect(ri.applyFilters(fake(), []).calls).toEqual([]);
|
|
});
|
|
});
|
|
|
|
describe('measure', () => {
|
|
const spec = { id: 'r1', source: 'f.js:1', table: 't', key: ['id'] };
|
|
|
|
it('assembles a full result for a clean reader', async () => {
|
|
const r = await ri.measure(spec, {
|
|
pageSize: 10,
|
|
exactCount: async () => 15,
|
|
fetchUnordered: pagesOf([ids(1, 10), ids(11, 15)]),
|
|
fetchOrdered: pagesOf([ids(1, 10), ids(11, 15)]),
|
|
});
|
|
expect(r.verdict).toBe(ri.VERDICT.PASS);
|
|
expect(r.pages).toBe(2);
|
|
expect(r.id).toBe('r1');
|
|
});
|
|
|
|
it('reports ERROR (not PASS, not empty) when a page fetch throws', async () => {
|
|
const r = await ri.measure(spec, {
|
|
pageSize: 10,
|
|
exactCount: async () => 15,
|
|
fetchUnordered: async () => { throw new Error('boom'); },
|
|
fetchOrdered: pagesOf([ids(1, 15)]),
|
|
});
|
|
expect(r.verdict).toBe(ri.VERDICT.ERROR);
|
|
expect(r.reason).toMatch(/boom/);
|
|
});
|
|
|
|
it('reports ERROR when the count itself fails', async () => {
|
|
const r = await ri.measure(spec, {
|
|
pageSize: 10,
|
|
exactCount: async () => { throw new Error('count failed'); },
|
|
fetchUnordered: pagesOf([ids(1, 5)]),
|
|
fetchOrdered: pagesOf([ids(1, 5)]),
|
|
});
|
|
expect(r.verdict).toBe(ri.VERDICT.ERROR);
|
|
});
|
|
});
|
|
|
|
describe('the harness writes nothing', () => {
|
|
it('has no insert/update/upsert/delete anywhere in its source', () => {
|
|
const fs = require('fs');
|
|
const path = require('path');
|
|
for (const f of ['../../src/utils/readIntegrity.js', '../../scripts/read-integrity.js']) {
|
|
const p = path.join(__dirname, f);
|
|
if (!fs.existsSync(p)) continue;
|
|
const src = fs.readFileSync(p, 'utf8');
|
|
expect(src).not.toMatch(/\.insert\(|\.update\(|\.upsert\(|\.delete\(|\.rpc\(/);
|
|
}
|
|
});
|
|
});
|