mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
feat(evals): per-case pass rates with Wilson intervals, identity series and quarantine policy
scripts/eval-flake-rank.ts becomes eval:pass-rates (eval:flake-rank stays an alias, and the legacy aggregate stays exported). It reads eval-store's trial-outcomes JSONL from the last N completed evals-periodic runs on this branch and main (gh, downloading only the trial-outcomes artifact, cached and size-capped, parsed as data), plus local eval dirs, and prints per-case per-trial pass rates with 95% Wilson intervals. A series is a case's own touchfiles minus GLOBAL_TOUCHFILES (caseSeriesIdentities, for the report job to stamp), per model, CLI version and policy version. Labels: INCONCLUSIVE, BROKEN, FLAKY, FAILING, PASSING. --backfill imports legacy slice artifacts as pre-policy trials (first attempt only, attributed by registry id, never guessed) for display only. --gate fails with ACTION REQUIRED on post-policy evidence only: drift below the quarantine entry rule, a rule case behaving like behavior, a one-sided Fisher drop against the previous identity (Holm-controlled), and quarantine entries that met their exit rule, expired after 8 weekly runs, broke the 10% tier cap or are invalid. CASE_QUARANTINE entries now carry a failureClass (detector, harness or model-latency); a product defect has no class and is never quarantined. The policy test pins EVAL_POLICY's approved constants.
This commit is contained in:
1 parent
0292ee3f6f
commit
d6b3559b78
4 files changed
+931
-44
No files matched your search
@@ -13,6 +13,13 @@ import * as path from 'node:path';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { aggregate, collectEvalFiles } from '../scripts/eval-flake-rank';
|
||||
import { manualReviewFixture } from './helpers/manual-judge-review-fixture';
|
||||
import {
|
||||
analyzePassRates, attributeLegacyRecord, backfillEvalFiles, caseSeriesIdentities, downloadRunArtifacts, fisherOneSidedLower,
|
||||
formatPassRates, holmRejections, listWeeklyRuns, quarantinePolicyProblems, quarantineRunsSince, readTrialOutcomeDir,
|
||||
wilsonInterval, type HistoryFetcher, type PassRatePolicy, type QuarantineEntry, type Registry, type TrialRecord,
|
||||
} from '../scripts/eval-flake-rank';
|
||||
import { EVAL_POLICY } from './helpers/periodic-exclude-data';
|
||||
import { TRIAL_OUTCOME_SCHEMA, formatTrialOutcomes } from './helpers/eval-store';
|
||||
|
||||
const entry = (name: string, passed: boolean, attempt: number) => ({
|
||||
name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1,
|
||||
@@ -39,9 +46,10 @@ describe('eval-flake-rank aggregate', () => {
|
||||
const display = spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), '--dir', dir],
|
||||
{ encoding: 'utf8', timeout: 10_000 });
|
||||
expect(display.status, display.stderr).toBe(0);
|
||||
expect(display.stdout).toContain('fails/runs manual');
|
||||
expect(display.stdout).toContain('0/1');
|
||||
expect(display.stdout).toContain(manual.name);
|
||||
// pass-rates view: the prior automated pass is the one scored pre-policy
|
||||
// trial; the manual acceptance is counted in its own column, never scored.
|
||||
expect(display.stdout).toContain('pre-policy manual case');
|
||||
expect(display.stdout).toMatch(new RegExp(`1/1 \\[[^\\]]+\\]\\s+1 ${manual.name.replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')}`));
|
||||
fs.writeFileSync(path.join(dir, 'invalid-retry.json'), run([
|
||||
{ ...ordinary, attempt: 1 }, { ...manual, attempt: 2 },
|
||||
]));
|
||||
@@ -97,3 +105,298 @@ describe('eval-flake-rank aggregate', () => {
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
});
|
||||
|
||||
// --- pass-rates ---
|
||||
|
||||
const registry: Registry = {
|
||||
kinds: { 'rule-a': 'rule', 'beh-b': 'behavior', 'gate-c': 'rule', 'mar-d': 'rule', 'judge one': 'judge',
|
||||
...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, 'rule'])) },
|
||||
tiers: { 'rule-a': 'periodic', 'beh-b': 'periodic', 'gate-c': 'gate', 'mar-d': 'marathon',
|
||||
...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, i < 9 ? 'gate' : 'periodic'])) },
|
||||
touchfiles: { 'rule-a': ['test/skill-e2e-a.test.ts', 'a/**'], 'beh-b': ['test/skill-e2e-b.test.ts', 'b/**'],
|
||||
'gate-c': ['test/skill-e2e-shared.test.ts'], 'mar-d': ['test/skill-e2e-shared.test.ts'] },
|
||||
judgeTouchfiles: { 'judge one': ['j/SKILL.md'] },
|
||||
globals: ['harness/**'],
|
||||
testNames: { 'gate-c': '/gate c labeled' },
|
||||
};
|
||||
|
||||
let clock = 0;
|
||||
function trial(id: string, outcome: 'passed' | 'failed' | 'skipped', extra: Partial<TrialRecord> = {}): TrialRecord {
|
||||
clock += 1;
|
||||
return {
|
||||
schema: TRIAL_OUTCOME_SCHEMA, case: id, file: 'test/x.test.ts', tier: registry.tiers[id] ?? 'judge',
|
||||
kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1, outcome,
|
||||
...(outcome === 'failed' ? { failure_class: 'assertion' as const } : {}),
|
||||
duration_ms: 1, cost_usd: 0, model: 'model-x', cli_version: '2.1.284', policy_version: 1, quarantined: false,
|
||||
execution: 'executed', source: 'shard', run_id: `run-${clock}`, recorded_at: new Date(Date.UTC(2026, 9, 1) + clock * 60_000).toISOString(),
|
||||
series_identity: 'id-1', ...extra,
|
||||
};
|
||||
}
|
||||
const many = (id: string, passes: number, fails: number, extra: Partial<TrialRecord> = {}) =>
|
||||
[...Array.from({ length: passes }, () => trial(id, 'passed', extra)), ...Array.from({ length: fails }, () => trial(id, 'failed', extra))];
|
||||
const analyze = (records: TrialRecord[], quarantine: Record<string, QuarantineEntry> = {}, extra = {}) =>
|
||||
analyzePassRates(records, { registry, quarantine, now: Date.UTC(2026, 9, 2), ...extra });
|
||||
const qEntry = (overrides: Partial<QuarantineEntry> = {}): QuarantineEntry => ({
|
||||
reason: 'Detector graded the posture wording; 7 of 10 fresh trials failed only the regex, transcripts attached.',
|
||||
failureClass: 'detector', tracking: 'TODOS.md "x"', owner: 'garrytan', enteredAt: '2026-09-29',
|
||||
exit: '>= 97% over >= 10 trials on the current identity', ...overrides,
|
||||
});
|
||||
|
||||
describe('pass-rates statistics', () => {
|
||||
test('Wilson bounds match the documented policy arithmetic', () => {
|
||||
expect(wilsonInterval(10, 10).lo).toBeCloseTo(0.7225, 4);
|
||||
expect(wilsonInterval(6, 6).lo).toBeCloseTo(0.6097, 4);
|
||||
expect(wilsonInterval(125, 125).lo).toBeCloseTo(0.9702, 4);
|
||||
expect(wilsonInterval(10, 10).hi).toBe(1);
|
||||
expect(wilsonInterval(0, 0)).toEqual({ lo: 0, hi: 1 });
|
||||
const mid = wilsonInterval(7, 10);
|
||||
expect(mid.lo).toBeGreaterThan(0.39); expect(mid.hi).toBeLessThan(0.9);
|
||||
});
|
||||
|
||||
test('one-sided Fisher exact matches a known table and is one-sided', () => {
|
||||
expect(fisherOneSidedLower(4, 6, 6, 6)).toBeCloseTo(0.227272727, 8);
|
||||
expect(fisherOneSidedLower(6, 6, 4, 6)).toBe(1);
|
||||
expect(fisherOneSidedLower(0, 10, 10, 10)).toBeLessThan(1e-4);
|
||||
});
|
||||
|
||||
test('Holm rejects step-down and stops at the first non-rejection', () => {
|
||||
expect([...holmRejections([0.001, 0.02, 0.04], 0.05)].sort()).toEqual([0, 1, 2]);
|
||||
expect([...holmRejections([0.001, 0.03, 0.04], 0.05)]).toEqual([0]);
|
||||
expect([...holmRejections([0.03, 0.04], 0.05)]).toEqual([]);
|
||||
expect([...holmRejections([], 0.05)]).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('pass-rates labels', () => {
|
||||
test('thin history is INCONCLUSIVE, and after this PR every series starts there', () => {
|
||||
const report = analyze(many('rule-a', 9, 0));
|
||||
expect(report.cases[0]).toMatchObject({ case: 'rule-a', label: 'INCONCLUSIVE' });
|
||||
expect(formatPassRates(analyze([]))).toContain('every series starts INCONCLUSIVE');
|
||||
});
|
||||
|
||||
test('PASSING, FLAKY and FAILING come from the interval against the entry rate', () => {
|
||||
expect(analyze(many('rule-a', 12, 0)).cases[0]!.label).toBe('PASSING');
|
||||
expect(analyze(many('beh-b', 10, 1)).cases[0]!.label).toBe('FLAKY');
|
||||
expect(analyze(many('beh-b', 2, 10)).cases[0]!.label).toBe('FAILING');
|
||||
});
|
||||
|
||||
test('BROKEN: the latest run is 0/n after a prior interval at or above the entry rate', () => {
|
||||
const prior = many('beh-b', 80, 0, { run_id: 'old' });
|
||||
const latest = [1, 2, 3].map(n => trial('beh-b', 'failed', { run_id: 'new', trial: n, panel: { n: 3, k: 2 } }));
|
||||
expect(analyze([...prior, ...latest]).cases[0]!.label).toBe('BROKEN');
|
||||
});
|
||||
|
||||
test('skipped trials carry no verdict; infra failures count as failed trials', () => {
|
||||
const stats = analyze([...many('rule-a', 10, 0), trial('rule-a', 'skipped'),
|
||||
trial('rule-a', 'failed', { failure_class: 'infra' })]).cases[0]!.current!;
|
||||
expect(stats).toMatchObject({ passes: 10, trials: 11, infra: 1 });
|
||||
});
|
||||
|
||||
test('a new identity, model or CLI starts a new series; earlier series stay visible', () => {
|
||||
const report = analyze([...many('rule-a', 10, 0), ...many('rule-a', 3, 0, { series_identity: 'id-2' }),
|
||||
...many('rule-a', 2, 0, { series_identity: 'id-2', cli_version: '2.1.285' })]);
|
||||
const c = report.cases[0]!;
|
||||
expect(c.series).toHaveLength(3);
|
||||
expect(c.current).toMatchObject({ identity: 'id-2', cli: '2.1.285', trials: 2 });
|
||||
expect(c.previous).toMatchObject({ identity: 'id-2', cli: '2.1.284', trials: 3 });
|
||||
expect(c.label).toBe('INCONCLUSIVE');
|
||||
});
|
||||
});
|
||||
|
||||
describe('pass-rates alarms count post-policy trials of the current series only', () => {
|
||||
test('backfilled pre-policy failures are displayed but never alarm', () => {
|
||||
const report = analyze(many('rule-a', 2, 20, { policy_version: 0, source: 'backfill' }));
|
||||
expect(report.alarms).toEqual([]);
|
||||
expect(report.cases[0]!.prePolicy).toMatchObject({ passes: 2, trials: 22 });
|
||||
expect(report.cases[0]!.label).toBe('INCONCLUSIVE');
|
||||
});
|
||||
|
||||
test('drift proposes quarantine for a blocking case below the entry rule; a rule case is flagged as behaving like behavior', () => {
|
||||
const kinds = analyze([...many('rule-a', 8, 2), ...many('beh-b', 8, 2), ...many('mar-d', 0, 10)]).alarms.map(a => `${a.kind}:${a.case}`);
|
||||
expect([...kinds].sort()).toEqual(['drift:beh-b', 'drift:rule-a', 'rule-as-behavior:mar-d', 'rule-as-behavior:rule-a']);
|
||||
expect(analyze(many('rule-a', 19, 1)).alarms).toEqual([]);
|
||||
});
|
||||
|
||||
test('the Fisher regression alarm needs the minimum trials on both sides', () => {
|
||||
const old = many('gate-c', 6, 0, { series_identity: 'old' });
|
||||
const fresh = many('gate-c', 0, 6, { series_identity: 'new' });
|
||||
expect(analyze([...old, ...fresh]).alarms.map(a => a.kind)).toContain('regression');
|
||||
expect(analyze([...old, ...fresh.slice(0, 5)]).alarms.map(a => a.kind)).not.toContain('regression');
|
||||
});
|
||||
|
||||
test('quarantine exit, expiry and cap', () => {
|
||||
const exit = analyze(many('beh-b', 10, 0), { 'beh-b': qEntry() }).alarms.map(a => a.kind);
|
||||
expect(exit).toContain('quarantine-exit');
|
||||
expect(exit).not.toContain('drift');
|
||||
const weekly = Array.from({ length: 8 }, (_, i) => new Date(Date.UTC(2026, 8, 30) + i * 7 * 86_400_000).toISOString());
|
||||
expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly }).alarms.map(a => a.kind)).toContain('quarantine-expired');
|
||||
expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly.slice(0, 7) }).alarms.map(a => a.kind)).not.toContain('quarantine-expired');
|
||||
expect(quarantineRunsSince('2026-09-01', undefined, Date.UTC(2026, 9, 27))).toBe(8);
|
||||
expect(quarantineRunsSince('not a date', undefined, 0)).toBe(Number.POSITIVE_INFINITY);
|
||||
});
|
||||
});
|
||||
|
||||
describe('quarantine policy', () => {
|
||||
const policy: PassRatePolicy = EVAL_POLICY;
|
||||
const now = Date.UTC(2026, 9, 2);
|
||||
test('a valid entry has no problems', () => {
|
||||
expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]);
|
||||
});
|
||||
|
||||
test('a product defect, a missing diagnosis or field, a bad date or a non-blocking case is rejected', () => {
|
||||
const problems = (quarantine: Record<string, QuarantineEntry>) => quarantinePolicyProblems(quarantine, registry, policy, now).map(p => p.message);
|
||||
expect(problems({ 'beh-b': qEntry({ failureClass: 'product' as QuarantineEntry['failureClass'] }) }).join()).toContain('never quarantined');
|
||||
expect(problems({ 'beh-b': qEntry({ reason: 'flaky' }) }).join()).toContain('written diagnosis');
|
||||
expect(problems({ 'beh-b': qEntry({ owner: ' ' }) }).join()).toContain('missing owner');
|
||||
expect(problems({ 'beh-b': qEntry({ enteredAt: '09/29/2026' }) }).join()).toContain('YYYY-MM-DD');
|
||||
expect(problems({ 'beh-b': qEntry({ enteredAt: '2027-01-01' }) }).join()).toContain('future');
|
||||
expect(problems({ 'mar-d': qEntry() }).join()).toContain('not blocking');
|
||||
expect(problems({ 'judge one': qEntry() }).join()).toContain('no registered E2E case');
|
||||
expect(problems({ ghost: qEntry() }).join()).toContain('no registered E2E case');
|
||||
});
|
||||
|
||||
test('at most 10% of a tier may be quarantined', () => {
|
||||
// 11 periodic cases in the fixture registry: the cap is 1.
|
||||
expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]);
|
||||
const over = quarantinePolicyProblems({ 'beh-b': qEntry(), 'rule-a': qEntry() }, registry, policy, now);
|
||||
expect(over.map(p => p.kind)).toEqual(['quarantine-cap']);
|
||||
expect(over[0]!.message).toContain('2 quarantined periodic cases exceed the 10% cap (1 of 11)');
|
||||
});
|
||||
});
|
||||
|
||||
describe('pass-rates inputs', () => {
|
||||
test('trial-outcomes JSONL is schema-validated; invalid lines are reported, never guessed', () => {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-'));
|
||||
const valid = trial('rule-a', 'passed');
|
||||
fs.mkdirSync(path.join(dir, 'nested'));
|
||||
fs.writeFileSync(path.join(dir, 'nested', 'trial-outcomes.jsonl'), formatTrialOutcomes([valid]) + '{"schema":"other"}\nnot json\n');
|
||||
fs.writeFileSync(path.join(dir, 'unrelated.jsonl'), formatTrialOutcomes([valid]));
|
||||
const read = readTrialOutcomeDir(dir);
|
||||
expect(read.records).toHaveLength(1);
|
||||
expect(read.records[0]).toMatchObject({ case: 'rule-a', series_identity: 'id-1' });
|
||||
expect(read.errors).toHaveLength(2);
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('legacy records attribute by shard suffix, id, label or single-owner file, else stay unattributed', () => {
|
||||
expect(attributeLegacyRecord('/anything', 'skill-e2e-b--beh-b', registry)).toBe('beh-b');
|
||||
expect(attributeLegacyRecord('rule-a', 'skill-e2e-zzz', registry)).toBe('rule-a');
|
||||
expect(attributeLegacyRecord('/gate c labeled', undefined, registry)).toBe('gate-c');
|
||||
expect(attributeLegacyRecord('/a display name', 'skill-e2e-a', registry)).toBe('rule-a');
|
||||
expect(attributeLegacyRecord('/shared display', 'skill-e2e-shared', registry)).toBeNull();
|
||||
});
|
||||
|
||||
test('backfill keeps only first attempts, defaults a missing attempt to 1, and labels records pre-policy', () => {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-backfill-'));
|
||||
fs.writeFileSync(path.join(dir, 'run.json'), run([
|
||||
{ ...entry_('rule-a', false, 1), exit_reason: 'timeout' }, entry_('rule-a', true, 2),
|
||||
{ name: 'beh-b', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 },
|
||||
entry_('/unknown display', true, 1),
|
||||
], { shard: 'skill-e2e-zzz' }));
|
||||
const { records, unattributed } = backfillEvalFiles(collectEvalFiles(dir), { run_id: '42', sha: 'abc' }, registry);
|
||||
expect(records.map(r => [r.case, r.outcome, r.failure_class, r.policy_version, r.source, r.run_id]))
|
||||
.toEqual([['rule-a', 'failed', 'timeout', 0, 'backfill', '42'], ['beh-b', 'passed', undefined, 0, 'backfill', '42']]);
|
||||
expect(formatTrialOutcomes(records)).toContain(TRIAL_OUTCOME_SCHEMA);
|
||||
expect(unattributed).toEqual(['/unknown display']);
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('series identity follows the case\'s own touchfiles, not GLOBAL_TOUCHFILES', () => {
|
||||
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-series-'));
|
||||
const git = (...args: string[]) => spawnSync('git', args, { cwd: root, encoding: 'utf8', timeout: 10_000 });
|
||||
for (const [file, body] of [['a/x.ts', '1'], ['b/y.ts', '1'], ['harness/run.ts', '1'], ['test/skill-e2e-a.test.ts', '1']] as const) {
|
||||
fs.mkdirSync(path.join(root, path.dirname(file)), { recursive: true });
|
||||
fs.writeFileSync(path.join(root, file), body);
|
||||
}
|
||||
const snapshot = () => { expect(git('add', '-A').status).toBe(0); return caseSeriesIdentities(['rule-a', 'beh-b'], root, registry); };
|
||||
expect(git('init', '-q').status).toBe(0);
|
||||
const first = snapshot();
|
||||
expect(first['rule-a']).not.toBe(first['beh-b']);
|
||||
fs.writeFileSync(path.join(root, 'harness/run.ts'), '2');
|
||||
expect(snapshot()).toEqual(first);
|
||||
fs.writeFileSync(path.join(root, 'a/x.ts'), '2');
|
||||
const next = snapshot();
|
||||
expect(next['rule-a']).not.toBe(first['rule-a']);
|
||||
expect(next['beh-b']).toBe(first['beh-b']);
|
||||
fs.rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
});
|
||||
|
||||
describe('pass-rates history fetch (injected, no network)', () => {
|
||||
function storedZip(files: Record<string, string>): Buffer {
|
||||
const locals: Buffer[] = [], centrals: Buffer[] = [];
|
||||
let offset = 0;
|
||||
for (const [name, text] of Object.entries(files)) {
|
||||
const data = Buffer.from(text), fileName = Buffer.from(name), crc = Bun.hash.crc32(data) >>> 0;
|
||||
const local = Buffer.alloc(30); local.writeUInt32LE(0x04034b50, 0); local.writeUInt16LE(20, 4);
|
||||
local.writeUInt32LE(crc, 14); local.writeUInt32LE(data.length, 18); local.writeUInt32LE(data.length, 22); local.writeUInt16LE(fileName.length, 26);
|
||||
const central = Buffer.alloc(46); central.writeUInt32LE(0x02014b50, 0); central.writeUInt16LE(20, 4); central.writeUInt16LE(20, 6);
|
||||
central.writeUInt32LE(crc, 16); central.writeUInt32LE(data.length, 20); central.writeUInt32LE(data.length, 24);
|
||||
central.writeUInt16LE(fileName.length, 28); central.writeUInt32LE(offset, 42);
|
||||
locals.push(local, fileName, data); centrals.push(central, fileName);
|
||||
offset += 30 + fileName.length + data.length;
|
||||
}
|
||||
const size = centrals.reduce((sum, b) => sum + b.length, 0);
|
||||
const end = Buffer.alloc(22); end.writeUInt32LE(0x06054b50, 0); end.writeUInt16LE(Object.keys(files).length, 8);
|
||||
end.writeUInt16LE(Object.keys(files).length, 10); end.writeUInt32LE(size, 12); end.writeUInt32LE(offset, 16);
|
||||
return Buffer.concat([...locals, ...centrals, end]);
|
||||
}
|
||||
|
||||
test('lists runs per branch, deduplicated and newest first', () => {
|
||||
const fetcher: HistoryFetcher = {
|
||||
listRuns: (_repo, _workflow, branch) => branch === 'main'
|
||||
? [{ id: 1, attempt: 1, sha: 'a', branch, createdAt: '2026-09-01T00:00:00Z' }, { id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }]
|
||||
: [{ id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }, { id: 2, attempt: 2, sha: 'b', branch, createdAt: '2026-09-08T00:00:00Z' }],
|
||||
listArtifacts: () => [], downloadZip: () => { throw new Error('unused'); },
|
||||
};
|
||||
expect(listWeeklyRuns({ repo: 'o/r', workflow: 'evals-periodic.yml', branches: ['feature', 'main'], limit: 10, fetcher }).map(r => r.id)).toEqual([3, 2, 1]);
|
||||
});
|
||||
|
||||
test('downloads only matching, bounded artifacts once, and caches them', () => {
|
||||
const cacheDir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cache-'));
|
||||
const downloads: number[] = [];
|
||||
const fetcher: HistoryFetcher = {
|
||||
listRuns: () => [],
|
||||
listArtifacts: () => [{ id: 10, name: 'trial-outcomes-gate', size: 100 }, { id: 11, name: 'paid-slice-1', size: 100 },
|
||||
{ id: 12, name: 'trial-outcomes-huge', size: 10 ** 9 }, { id: 13, name: 'trial-outcomes/../escape', size: 1 }],
|
||||
downloadZip: (_repo, id, destination) => { downloads.push(id); fs.writeFileSync(destination, storedZip({ 'trial-outcomes.jsonl': formatTrialOutcomes([trial('rule-a', 'passed')]) })); },
|
||||
};
|
||||
const options = { repo: 'o/r', run: { id: 7, attempt: 1, sha: 's', branch: 'main', createdAt: '' }, cacheDir, fetcher,
|
||||
match: (name: string) => name.startsWith('trial-outcomes') };
|
||||
const dirs = downloadRunArtifacts(options);
|
||||
expect(downloads).toEqual([10]);
|
||||
expect(dirs).toHaveLength(1);
|
||||
expect(readTrialOutcomeDir(dirs[0]!).records.map(r => r.case)).toEqual(['rule-a']);
|
||||
expect(downloadRunArtifacts(options)).toEqual(dirs);
|
||||
expect(downloads).toEqual([10]);
|
||||
fs.rmSync(cacheDir, { recursive: true, force: true });
|
||||
});
|
||||
});
|
||||
|
||||
describe('pass-rates CLI', () => {
|
||||
const cli = (args: string[]) => spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), ...args],
|
||||
{ encoding: 'utf8', timeout: 20_000 });
|
||||
|
||||
test('--dir prints per-case pass rates; --gate fails only on ACTION REQUIRED', () => {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cli-'));
|
||||
const id = 'plan-ceo-review-format-mode';
|
||||
const records = Array.from({ length: 12 }, (_, i) => ({ ...trial(id, i < 11 ? 'passed' : 'failed'), kind: 'behavior' as const, tier: 'periodic' }));
|
||||
fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records));
|
||||
const shown = cli(['--dir', dir, '--case', id]);
|
||||
expect(shown.status, shown.stderr).toBe(0);
|
||||
expect(shown.stdout).toContain(`11/12 [`);
|
||||
expect(shown.stdout).toMatch(new RegExp(`FLAKY\\s+behavior\\s+periodic.*${id}`));
|
||||
expect(shown.stdout).toContain('ACTION REQUIRED');
|
||||
expect(shown.stdout).toContain(`[drift] ${id} passes 11/12`);
|
||||
expect(cli(['--dir', dir, '--gate']).status).toBe(1);
|
||||
fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records.slice(0, 11)));
|
||||
const clean = cli(['--dir', dir, '--gate', '--json']);
|
||||
expect(clean.status, clean.stdout).toBe(0);
|
||||
expect(JSON.parse(clean.stdout).cases[0]).toMatchObject({ case: id, label: 'PASSING', current: { passes: 11, trials: 11 } });
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
});
|
||||
|
||||
function entry_(name: string, passed: boolean, attempt: number) {
|
||||
return { name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1 };
|
||||
}
|
||||
@@ -106,14 +106,18 @@ export const EVAL_POLICY = {
|
||||
* failures (a product defect is never quarantined), and unchanged case
|
||||
* touchfiles in the change that adds it. Pinned by
|
||||
* test/periodic-exclude-policy.test.ts.
|
||||
* reason - the written diagnosis
|
||||
* tracking - issue or TODOS pointer
|
||||
* owner - who removes it
|
||||
* enteredAt - ISO date the entry landed (expiry counts weekly runs from here)
|
||||
* exit - the measurable exit condition
|
||||
* reason - the written diagnosis, with the pass-rate evidence
|
||||
* failureClass - what the diagnosis found; a product defect has no class here
|
||||
* tracking - issue or TODOS pointer
|
||||
* owner - who removes it
|
||||
* enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here)
|
||||
* exit - the measurable exit condition
|
||||
* At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and
|
||||
* periodic are the blocking tiers) may be quarantined at once.
|
||||
*/
|
||||
export const CASE_QUARANTINE: Record<string, {
|
||||
reason: string;
|
||||
failureClass: 'detector' | 'harness' | 'model-latency';
|
||||
tracking: string;
|
||||
owner: string;
|
||||
enteredAt: string;
|
||||
|
||||
@@ -9,10 +9,11 @@ import { describe, expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
|
||||
import { CASE_CI_EXCLUDE, PERIODIC_CI_EXCLUDE } from './helpers/periodic-exclude-data';
|
||||
import { CASE_CI_EXCLUDE, CASE_QUARANTINE, EVAL_POLICY, PERIODIC_CI_EXCLUDE } from './helpers/periodic-exclude-data';
|
||||
import { E2E_TOUCHFILES } from './helpers/touchfiles';
|
||||
import { quarantinePolicyProblems } from '../scripts/eval-flake-rank';
|
||||
import { isPaidTestFile } from './helpers/paid-test-set';
|
||||
import { buildRunManifest, CASE_SHARDED_FILES, expandCaseShards, partitionCaseExclusions, selectPaidTestFiles, shardCaseId, shardFile } from '../scripts/test-paid-shards';
|
||||
import { buildRunManifest, CASE_SHARDED_FILES, expandCaseShards, fileCaseRegistration, partitionCaseExclusions, selectPaidTestFiles, shardCaseId, shardFile } from '../scripts/test-paid-shards';
|
||||
|
||||
const ROOT = path.resolve(__dirname, '..');
|
||||
|
||||
@@ -72,3 +73,30 @@ describe('periodic exclude policy', () => {
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('eval verdict policy (pre-registered)', () => {
|
||||
test('EVAL_POLICY carries exactly the approved constants; a change needs re-approval and a version bump', () => {
|
||||
expect(EVAL_POLICY).toEqual({
|
||||
version: 1,
|
||||
panel: { n: 3, k: 2 },
|
||||
quarantine: { entry: { rate: 0.95, minTrials: 10 }, exit: { rate: 0.97, minTrials: 10 }, capFraction: 0.10, expiryWeeklyRuns: 8 },
|
||||
judge: { samples: 3 },
|
||||
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
|
||||
infraRedispatch: 1,
|
||||
});
|
||||
});
|
||||
|
||||
test('every CASE_QUARANTINE entry is a diagnosed, dated, non-product blocking case within the tier cap', () => {
|
||||
expect(quarantinePolicyProblems(CASE_QUARANTINE).map(problem => problem.message)).toEqual([]);
|
||||
});
|
||||
|
||||
test('a quarantined case runs as isolated trial shards: its files register it literally', () => {
|
||||
for (const id of Object.keys(CASE_QUARANTINE)) {
|
||||
const files = (E2E_TOUCHFILES[id] ?? []).filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file));
|
||||
expect(files.length, `${id}: no paid file registers it`).toBeGreaterThan(0);
|
||||
for (const file of files) {
|
||||
expect(fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')).known, `${id}: ${file} registration must be statically known`).toBe(true);
|
||||
}
|
||||
}
|
||||
});
|
||||
});
|
||||
Reference in new issue
Block a user