fix(evals): stop the harness grading itself

findPreviousRun excluded only the file being written, by name, so every
suite compared against _partial-e2e.json — the current run's own
accumulator, relabelled with the current tier just before each flush.
That is why every block read '+$0.00, +0s, Stable run, no regressions.'
This harness has never been able to detect a regression, and reassuring
output that cannot fail is worse than none. In-progress runs are now
excluded by role, and a run with nothing to compare against says NO
BASELINE instead of claiming stability.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
(cherry picked from commit f3140b5245221fff7fb9411c7ec07c2ca11587b5)
This commit is contained in:
Sinabina
2026-08-12 15:31:47 -07:00
committed by Garry Tan
parent e78db4e309
commit 39e8e789bf
2 changed files with 129 additions and 3 deletions
+99
View File
@@ -58,6 +58,19 @@ function makeResult(overrides?: Partial<EvalResult>): EvalResult {
};
}
/** Capture everything a block writes to stderr (finalize prints there). */
async function captureStderr(fn: () => Promise<void>): Promise<string> {
const original = process.stderr.write.bind(process.stderr);
let captured = '';
(process.stderr as any).write = (chunk: any) => { captured += String(chunk); return true; };
try {
await fn();
} finally {
(process.stderr as any).write = original;
}
return captured;
}
// --- EvalCollector tests ---
describe('EvalCollector', () => {
@@ -119,6 +132,41 @@ describe('EvalCollector', () => {
expect(fs.readdirSync(tmpDir).filter(f => f.endsWith('.json') && !f.startsWith('_partial'))).toHaveLength(1);
});
test('with no completed prior run, says NO BASELINE instead of comparing against its own partial', async () => {
// addTest writes the in-progress accumulator into the same dir. If that
// counted as a baseline, the run would compare against itself and print a
// reassuring all-clear forever.
const collector = new EvalCollector('e2e', tmpDir);
collector.addTest(makeEntry({ name: 'test-1', passed: true }));
const output = await captureStderr(async () => { await collector.finalize(); });
expect(fs.existsSync(path.join(tmpDir, '_partial-e2e.json'))).toBe(true); // the trap exists
expect(output).toContain('NO BASELINE');
expect(output).not.toContain('vs previous');
expect(output).not.toContain('Stable run');
});
test('with a genuine prior run, reports the real delta', async () => {
fs.writeFileSync(
path.join(tmpDir, '0.3.5-main-e2e-20260312-100000.json'),
JSON.stringify(makeResult({
timestamp: '2026-03-12T10:00:00Z',
tests: [makeEntry({ name: 'test-1', passed: true, turns_used: 5 })],
})),
);
const collector = new EvalCollector('e2e', tmpDir);
collector.addTest(makeEntry({ name: 'test-1', passed: false, turns_used: 5 }));
const output = await captureStderr(async () => { await collector.finalize(); });
expect(output).toContain('vs previous');
expect(output).toContain('REGRESSION');
expect(output).toContain('1 regressed');
expect(output).not.toContain('NO BASELINE');
});
test('empty collector writes valid file', async () => {
const collector = new EvalCollector('llm-judge', tmpDir);
const filepath = await collector.finalize();
@@ -259,6 +307,34 @@ describe('findPreviousRun', () => {
expect(result).toBeNull(); // only file is excluded
});
test('never returns the in-progress accumulator as a baseline', () => {
// The current run's own partial carries the current tier + branch and the
// freshest timestamp. If it were a candidate, every run would compare
// against itself and report "no regressions" forever.
fs.writeFileSync(
path.join(tmpDir, '_partial-e2e.json'),
JSON.stringify(makeResult({ branch: 'main', timestamp: '2026-03-14T10:00:00Z', _partial: true })),
);
const result = findPreviousRun(tmpDir, 'e2e', 'main', path.join(tmpDir, 'current.json'));
expect(result).toBeNull();
});
test('prefers a completed run over a newer in-progress accumulator', () => {
fs.writeFileSync(
path.join(tmpDir, '0.3.5-main-e2e-20260312-100000.json'),
JSON.stringify(makeResult({ branch: 'main', timestamp: '2026-03-12T10:00:00Z' })),
);
// Newer, same tier + branch, but in-progress — must lose to the older completed run.
fs.writeFileSync(
path.join(tmpDir, '_partial-e2e.json'),
JSON.stringify(makeResult({ branch: 'main', timestamp: '2026-03-14T10:00:00Z', _partial: true })),
);
const result = findPreviousRun(tmpDir, 'e2e', 'main', path.join(tmpDir, 'current.json'));
expect(result).toContain('0.3.5-main-e2e');
});
test('filters by tier', () => {
fs.writeFileSync(
path.join(tmpDir, '0.3.6-main-llm-judge-20260314-100000.json'),
@@ -524,6 +600,29 @@ describe('generateCommentary', () => {
expect(notes.some(n => n.includes('No regressions'))).toBe(true);
});
test('says NO BASELINE instead of "stable" when nothing matched the prior run', () => {
// A baseline file existed but shares no test names (renamed/retired suite),
// so zero tests were actually compared. Claiming stability here is a lie.
const c: ComparisonResult = {
before_file: 'a.json', after_file: 'b.json',
before_branch: 'main', after_branch: 'main',
before_timestamp: '', after_timestamp: '',
deltas: [
{ name: 'a', before: { passed: false, cost_usd: 0 }, after: { passed: true, cost_usd: 0.10 }, status_change: 'unchanged' },
{ name: 'b', before: { passed: false, cost_usd: 0 }, after: { passed: true, cost_usd: 0.10 }, status_change: 'unchanged' },
{ name: 'c', before: { passed: false, cost_usd: 0 }, after: { passed: true, cost_usd: 0.10 }, status_change: 'unchanged' },
],
total_cost_delta: 0.30, total_duration_delta: 0,
improved: 0, regressed: 0, unchanged: 3,
tool_count_before: 0, tool_count_after: 0,
matched: 0,
};
const notes = generateCommentary(c);
expect(notes.some(n => n.includes('NO BASELINE'))).toBe(true);
expect(notes.some(n => n.includes('Stable run'))).toBe(false);
});
test('returns empty for stable run with no significant changes', () => {
const c: ComparisonResult = {
before_file: 'a.json', after_file: 'b.json',