Merge remote-tracking branch 'origin/capy/rel-a' into capy/rel-c

This commit is contained in:
garrytan committed 2026-09-29 19:47:17 +00:00
commit b2ca207cf0
46 files changed
+4936 -936

No files matched your search

+1 -1
View File
@@ -22,7 +22,7 @@ describe('carved-skill cases each get a complete paid process budget', () => {
expect(selectPaidTestFiles(files.map(file => 'test/' + file), 'periodic').selected).toHaveLength(files.length);
expect(selectPaidTestFiles(files.map(file => 'test/' + file), 'gate').selected).toHaveLength(0);
});
test('all configured retries plus teardown fit even with within-shard concurrency one', () => {
test('every case run plus teardown fits even with within-shard concurrency one', () => {
for (const file of files) {
const attempts = retriesForFiles(['test/' + file]) + 1;
expect(CAPTURE_LONG_MS * attempts + 10_000).toBeLessThan(DEFAULT_SHARD_TIMEOUT_MS);
+42 -56
View File
@@ -21,19 +21,31 @@ test('only PR runs select the fast profile; manual and scheduled coverage stays
}
});
test('receipt transport restores only this repository and PR with no broad fallback key', () => {
const steps = paid.jobs['eval-slices'].steps;
const restore = steps.filter((s: any) => s.uses?.startsWith('actions/cache/restore@'));
const save = steps.filter((s: any) => s.uses?.startsWith('actions/cache/save@'));
test('receipt transport: the planner restores only this repository and PR, the report saves one merged store', () => {
const planner = paid.jobs['plan-slices'].steps;
const restore = planner.filter((s: any) => s.uses?.startsWith('actions/cache/restore@'));
expect(restore).toHaveLength(1);
expect(save).toHaveLength(1);
expect(restore[0].if).toBe("github.event_name == 'pull_request'");
expect(restore[0].with.path).toBe('/tmp/gstack-eval-input-cache');
expect(restore[0].with['restore-keys']).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-');
expect(save[0].with.key).toBe(restore[0].with.key);
expect(save[0].with.key).toContain('${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}');
const emit = planner.find((s: any) => s.run?.includes('--emit-plan /tmp/paid-plan/manifest.json'));
expect(emit.env.EVALS_CACHE_DIR).toBe("${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }}");
const upload = planner.find((s: any) => s.with?.name === 'paid-plan');
expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/paid-plan/manifest.json', '/tmp/paid-plan/receipts']);
// Executors never restore or save a cache of their own: every slice sees the plan's one receipt set.
const executor = paid.jobs['eval-slices'].steps;
expect(executor.filter((s: any) => s.uses?.startsWith('actions/cache/'))).toHaveLength(0);
expect(executor.find((s: any) => s.name === "Seed this slice's receipts from the plan").run).toContain('cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/');
const report = paid.jobs['slices-report'].steps;
const merge = report.find((s: any) => s.name === "Merge this run's receipts");
expect(merge.run).toContain('scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache');
const save = report.filter((s: any) => s.uses?.startsWith('actions/cache/save@'));
expect(save).toHaveLength(1);
expect(save[0].with.path).toBe('/tmp/gstack-eval-input-cache');
expect(save[0].if).toContain("steps.receipts.outputs.present == 'true'");
expect(save[0].with.key).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged');
expect(report.indexOf(save[0])).toBeGreaterThan(report.indexOf(merge));
expect(paid.jobs['eval-slices'].permissions).toEqual({ contents: 'read', packages: 'read' });
expect(paid.jobs['slices-report'].permissions).toEqual({ contents: 'read' });
expect(JSON.stringify(periodic)).not.toContain('actions/cache/');
});
@@ -43,56 +55,21 @@ test('the judge binds cache receipts to the PR and installed runtime, not the co
expect(runtime.run).toContain('sha256sum /tmp/eval-runtime-manifest.json');
const run = paid.jobs['eval-slices'].steps.find((s: any) => s.run?.includes('--plan /tmp/paid-plan/manifest.json'));
expect(run.env).toMatchObject({
EVALS_CACHE_DIR: '/tmp/gstack-eval-input-cache',
EVALS_CACHE_DIR: '/tmp/paid-slice-results/receipts',
EVALS_CACHE_REPOSITORY: '${{ github.repository }}',
EVALS_CACHE_PR: '${{ github.event.pull_request.number }}',
EVALS_CACHE_RUNTIME_ID: '${{ needs.build-image.outputs.runtime-id }}',
});
});
test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('only a new passing producer can publish the next cache snapshot', () => {
const directory = mkdtempSync(join(tmpdir(), 'ci-cache-producer-'));
const receipts = join(directory, 'receipts');
const output = join(directory, 'output');
mkdirSync(receipts);
const step = paid.jobs['eval-slices'].steps.find((s: any) => s.id === 'receipts');
const script = step.run.replaceAll('/tmp/gstack-eval-input-cache', receipts);
const run = () => {
writeFileSync(output, '');
const result = spawnSync('bash', ['-e', '-c', script], {
env: { ...process.env, GITHUB_OUTPUT: output, GITHUB_RUN_ID: '42', GITHUB_RUN_ATTEMPT: '2' },
encoding: 'utf8', timeout: 5000,
});
expect(result.status, result.stderr).toBe(0);
return readFileSync(output, 'utf8');
};
try {
expect(run()).toBe('');
writeFileSync(join(receipts, 'old.json'), JSON.stringify({ proof: { source: { runId: '41/1' } } }));
writeFileSync(join(receipts, 'corrupt.json'), '{');
expect(run()).toBe('');
writeFileSync(join(receipts, 'prior-attempt.json'), JSON.stringify({ proof: { source: { runId: '42/1' } } }));
expect(run()).toBe('');
writeFileSync(join(receipts, 'fresh.json'), JSON.stringify({ proof: { source: { runId: '42/2' } } }));
expect(run()).toBe('present=true\n');
} finally { rmSync(directory, { recursive: true, force: true }); }
});
test.skipIf(!Bun.which('jq'))('the actual comment separates reused evidence, retry outcomes and deferred coverage', () => {
test.skipIf(!Bun.which('jq'))('the actual comment shows deferred coverage and never recomputes a verdict', () => {
const comment = paid.jobs['slices-comment'].steps.find((s: any) => s.name === 'Post PR comment').run as string;
const evaluate = (filter: string, value: unknown) => {
const result = spawnSync('jq', ['-r', filter], { input: JSON.stringify(value), encoding: 'utf8', timeout: 5000 });
expect(result.status, result.stderr).toBe(0);
return result.stdout.trim();
};
const stats = comment.match(/STATS=\$\(jq -r '([^']+)'/)![1]!;
expect(evaluate(stats, { tests: [
{ name: 'retry', passed: false }, { name: 'retry', passed: true },
{ name: 'exhausted', passed: false }, { name: 'exhausted', passed: false },
{ name: 'regressed', passed: true }, { name: 'regressed', passed: false },
{ name: 'reused', passed: true, execution: 'reused' },
], flaky_retries: ['retry', 'exhausted', 'regressed'].map(name => ({ name, attempts: 2 })) })).toBe('4 2 2 3 3 1');
expect(comment).toContain("printf ' | ⚠ %s cases with multiple attempts'");
expect(comment).not.toContain('group_by(.name)');
expect(comment).not.toMatch(/flaky pass\(es\)|passed only on retry|not blocking/);
const coverage = comment.match(/COVERAGE=\$\(jq -r '([^']+)'/)![1]!;
const text = evaluate(coverage, { profile: 'pr', selection: { e2e: ['probe'], judges: ['judge'] },
@@ -107,9 +84,9 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
const job = paid.jobs['slices-comment'];
expect(job.permissions).toMatchObject({ 'pull-requests': 'write' });
expect(JSON.stringify(job.steps)).not.toMatch(/actions\/checkout|setup-bun|bun run|npm |node /);
const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict');
expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json']);
expect(job.steps.find((step: any) => step.with?.name === 'report-verdict').with.path).toBe('/tmp/verdict');
const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}');
expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json', '/tmp/paid-report/report-summary.md']);
expect(job.steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}').with.path).toBe('/tmp/verdict');
const root = mkdtempSync(join(tmpdir(), 'ci-comment-'));
const paidDir = join(root, 'paid-report');
const verdictDir = join(root, 'verdict');
@@ -123,9 +100,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
writeFileSync(join(paidDir, 'judge.json'), JSON.stringify({ total_tests: 2, tier: 'llm-judge', shard: 1,
tests: [{ name: 'manual', passed: false, manual_review: { unverified: true } },
{ name: 'reused', passed: true, execution: 'reused' }], flaky_retries: [] }));
const summary = { version: 1, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0,
const summary = { version: 2, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0,
total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 }],
totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 } };
totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 },
verdict: { verdict: 'GREEN' }, headline: ['[test:paid] VERDICT GREEN — lane gate/pr, attempt 1'], panels: [],
failures: ['⚠ case-x behavior PASS 2/3 (✓✗✓) t2: timeout at turn 3 — @\u200bsomeone said no'] };
mkdirSync(join(verdictDir, 'paid-report'));
const summaryPath = join(verdictDir, 'paid-report/collector-outcomes.json');
const script = (job.steps.find((step: any) => step.name === 'Post PR comment').run as string)
@@ -142,8 +121,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
const verified = run();
expect(verified.status, verified.stderr).toBe(0);
expect(verified.stdout).toContain('⚠ MANUAL ACCEPTED (unscored)');
expect(verified.stdout).toContain('1 automated passed / 2 final results');
expect(verified.stdout).toContain('0 failed, 1 manual accepted');
expect(verified.stdout).toContain('VERDICT GREEN — lane gate/pr, attempt 1');
expect(verified.stdout).toContain('1 executed, 1 reused** rule/judge records');
expect(verified.stdout).toContain('1 manual accepted');
expect(verified.stdout).toContain('### Failures and split verdicts');
expect(verified.stdout).toContain('PASS 2/3 (✓✗✓) t2: timeout at turn 3');
const unrelatedFailure = { ...summary, files: [{ ...summary.files[0], total: 3, failed: 1,
executed: 2, attempts: 3 }], totals: { ...summary.totals, total: 3, failed: 1,
@@ -152,16 +134,20 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
const red = run();
expect(red.status, red.stderr).toBe(0);
expect(red.stdout).toContain('❌ FAIL');
expect(red.stdout).toContain('1 failed, 1 manual accepted');
writeFileSync(summaryPath, JSON.stringify({ ...summary, verdict: { verdict: 'RED' } }));
const redVerdict = run();
expect(redVerdict.status, redVerdict.stderr).toBe(0);
expect(redVerdict.stdout).toContain('❌ FAIL');
writeFileSync(summaryPath, JSON.stringify({ ...summary, totals: { ...summary.totals, manual_accepted: 2 } }));
const tampered = run();
expect(tampered.status, tampered.stderr).toBe(0);
expect(tampered.stdout).toContain('manual acceptance unavailable/unverified');
expect(tampered.stdout).toContain('verified report unavailable');
expect(tampered.stdout).not.toContain('⚠ MANUAL ACCEPTED (unscored)');
rmSync(summaryPath);
const absent = run();
expect(absent.status, absent.stderr).toBe(0);
expect(absent.stdout).toContain('manual acceptance unavailable/unverified');
expect(absent.stdout).toContain('verified report unavailable');
} finally { rmSync(root, { recursive: true, force: true }); }
});
+14 -6
View File
@@ -3,7 +3,7 @@ import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { buildRunManifest, collectPaidTestFiles, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards';
import { buildRunManifest, collectPaidTestFiles, shardCaseId, shardTrial, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards';
import { STRICT_RETRY_CASE_BUDGETS } from './helpers/eval-budgets';
import { approvedCookieWorkflowSource, manualReviewFixture } from './helpers/manual-judge-review-fixture';
@@ -16,6 +16,11 @@ type Job = {
permissions: Record<string, string>;
steps: Step[];
};
/** A passing trial record for an isolated trial shard (the executor's current result schema). */
const trialRecord = (entry: PaidRunManifest['entries'][number]) => entry.trial ? { trial: {
case: shardCaseId(entry.file)!, trial: shardTrial(entry.file)!, ...entry.trial, outcome: 'passed' as const, cost_usd: 0, duration_ms: 1,
} } : {};
const workflows = ['evals.yml', 'evals-periodic.yml'].map(name => ({
name,
jobs: (Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows', name), 'utf8')) as {
@@ -112,10 +117,11 @@ describe('paid CI coordination stays off the eval image', () => {
if (name === 'evals.yml') expect(report.permissions).toEqual({ contents: 'read' });
});
test(`${name}: failure logs include the hidden spool directory without uploading the rest of the cache`, () => {
const logs = jobs['eval-slices'].steps.find(step => step.with?.name === 'paid-slice-${{ matrix.slice }}-logs');
test(`${name}: shard logs include the hidden spool directory without uploading the rest of the cache`, () => {
const logs = jobs['eval-slices'].steps.find(step => step.with?.name === 'paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}');
expect(logs?.uses).toStartWith('actions/upload-artifact@');
expect(logs?.if).toBe('failure()');
// A failed trial no longer reds its runner; its log is still the evidence.
expect(logs?.if).toBe('always()');
expect(logs?.with?.['include-hidden-files']).toBe(true);
expect(String(logs?.with?.path).trim().split('\n')).toEqual([
'/home/runner/.cache/gstack-paid-shard-*.log',
@@ -217,6 +223,7 @@ describe('dependency-free CI planner and report execution', () => {
executedTests: STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === entry.file)?.cases ?? 1,
skippedTests: 0,
...(entry.budget ? { budget: entry.budget } : {}),
...trialRecord(entry),
})),
};
fs.writeFileSync(path.join(reportDir, `slice-${sliceIndex}.json`), JSON.stringify(result));
@@ -248,7 +255,8 @@ describe('dependency-free CI planner and report execution', () => {
const red = run(['--report', reportDir], tier);
expect(red.status).toBe(1);
expect(red.stderr).toContain(`${failed.outcomes[0].files[0]}: failed`);
expect(red.stdout).toContain('3 executed, 0 reused; 1 passed, 2 failed, 0 manual accepted (unscored; no score-cache credit) (6 attempt records from 1 collectors)');
// Paid evals never retry: every record counts, a later pass never hides an earlier failure.
expect(red.stdout).toContain('6 executed, 0 reused; 2 passed, 4 failed, 0 manual accepted (unscored; no score-cache credit) (6 attempt records from 1 collectors');
expect(red.stdout).toContain('3 cases with multiple attempts this run:');
expect(red.stdout).not.toMatch(/passed only on retry|not blocking/);
@@ -274,7 +282,7 @@ describe('dependency-free CI planner and report execution', () => {
outcomes: manifest.entries.filter(entry => entry.status === 'planned').map(entry => ({
files: [entry.file], status: 'passed', exitCode: 0, elapsedMs: 1,
executedTests: STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === entry.file)?.cases ?? 1,
skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}),
skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}), ...trialRecord(entry),
})),
};
const slicePath = path.join(reportDir, 'slice-1.json');
+7 -7
View File
@@ -11,7 +11,7 @@ import { selectTests } from './helpers/test-selection';
import { E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data';
import { selectPrProfile } from '../scripts/test-pr-profile';
import { JUDGE_MS } from './helpers/eval-budgets';
import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge';
import { COOKIE_MANUAL_REVIEW_FILE, getCookieWorkflowManualReview, isManualReviewEntry } from './helpers/cookie-workflow-manual-review';
const ROOT = resolve(import.meta.dir, '..');
@@ -70,7 +70,7 @@ function actualCookieCallback(root: string, overrides: {
const records: EvalTestEntry[] = [];
const attempts = new Map<string, { attempt: number }>();
let callback: () => Promise<void> = async () => { throw new Error('Judge callback was not registered'); };
new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', registration)(
new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', 'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS', registration)(
(_suite: string, names: string[], run: () => void) => { expect(names).toEqual([NAME]); run(); },
(name: string, run: () => Promise<void>, budget: number) => { expect(name).toBe(NAME); expect(budget).toBe(JUDGE_MS + 10_000); callback = run; },
root, buildCookieWorkflowJudgeInput, (_kind: string, explicit?: string) => explicit ?? 'fixture-model',
@@ -85,7 +85,7 @@ function actualCookieCallback(root: string, overrides: {
attempts, overrides.clock ? { now: overrides.clock } : performance,
overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout,
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS,
);
return { run: () => callback(), requests, records, attempts };
}
@@ -96,7 +96,7 @@ describe('cookie workflow judge input', () => {
approveFixture(root);
const h = actualCookieCallback(root, { judge: async () => { throw refusal(); } });
await h.run();
expect(h.requests).toHaveLength(1);
expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
expect(h.records).toHaveLength(1);
expect(h.records[0]).toMatchObject({ passed: false, execution: 'executed', exit_reason: 'provider_refusal' });
expect(isManualReviewEntry(h.records[0])).toBe(true);
@@ -161,7 +161,7 @@ describe('cookie workflow judge input', () => {
const root = fixture(); approveFixture(root);
let calls = 0;
const h = actualCookieCallback(root, { judge: async () => {
if (++calls === 1) return { ...passingScore, clarity: 1 };
if (++calls <= JUDGE_PANEL_SAMPLES) return { ...passingScore, clarity: 1 };
throw refusal();
} });
await expect(h.run()).rejects.toThrow();
@@ -275,7 +275,7 @@ describe('cookie workflow judge input', () => {
let scores = passingScore;
const h = actualCookieCallback(root, { judge: async () => scores });
await h.run();
expect(h.requests).toHaveLength(1);
expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
expect(h.requests[0].prompt).toBe(input.prompt);
expect(h.requests[0].model).toBe(COOKIE_WORKFLOW_JUDGE.model);
expect(h.requests[0].signal).toBeInstanceOf(AbortSignal);
@@ -283,7 +283,7 @@ describe('cookie workflow judge input', () => {
expect(existsSync(join(root, 'cache'))).toBe(false);
const fresh = actualCookieCallback(root);
await fresh.run();
expect(fresh.requests).toHaveLength(1);
expect(fresh.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
for (const dimension of ['clarity', 'completeness', 'actionability'] as const) {
scores = { ...COOKIE_WORKFLOW_JUDGE.thresholds, [dimension]: COOKIE_WORKFLOW_JUDGE.thresholds[dimension] - 1, reasoning: 'Synthetic failing fixture score' };
await expect(h.run()).rejects.toThrow();
+76 -2
View File
@@ -4,7 +4,8 @@ import * as os from 'node:os';
import * as path from 'node:path';
import {
e2eReuseEnvironment, e2eReuseLaneProblem, e2eShardIdentity, e2eShardInputFiles, prepareE2EShardReuse,
type E2EShardReuseRequest,
mergeReceiptDirs, readPanelReceipt, selectPlanReceipts, writeNegativeReceipt, writePanelReceipt,
type E2EShardReuseRequest, type PanelReceipt,
} from '../scripts/e2e-shard-reuse';
import { buildRunManifest, fileCaseRegistration, runPaidShard, verifySliceResults, type SliceResult } from '../scripts/test-paid-shards';
@@ -127,10 +128,13 @@ describe('E2E shard reuse through the runner', () => {
test('a failed shard never publishes a receipt', async () => {
let published = 0;
const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, env: laneEnv(), log: () => {},
reuseFor: () => ({ lookup: () => null, publish: () => { published++; } }),
reuseFor: () => ({ inputKey: 'e'.repeat(64), unchanged: () => true, lookupPanelTrial: () => null,
lookup: () => null, publish: () => { published++; } }),
commandFor: () => ({ command: process.execPath, args: ['-e', 'process.exit(1)'] }) });
expect(outcome.status).toBe('failed');
expect(published).toBe(0);
// The identity rides on the outcome so the report can store the FAIL as a negative receipt.
expect(outcome.inputKey).toBe('e'.repeat(64));
});
test('the report accepts reused results only in the fast PR profile', () => {
@@ -143,3 +147,73 @@ describe('E2E shard reuse through the runner', () => {
expect(verifySliceResults(manifest, results).problems).toContain(`${FILE}: only the fast PR profile may reuse results; this lane executes fresh`);
});
});
describe('planner-side panel reuse and negative receipts', () => {
const panelPlan = { kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined: false };
const trialRequest = (trial: number, over: Partial<E2EShardReuseRequest> = {}) => request({
key: `${FILE}#setup-deploy-workflow~t${trial}`, panel: panelPlan, ...over });
const source = (completedAt: number, runId = '1001/1') => ({ runId, revision: 'd'.repeat(40), completedAt });
const panel = (key: string, outcomes: Array<'passed' | 'failed'>, completedAt = Date.now() - 1_000): PanelReceipt => ({
schema: 1, key, case: 'setup-deploy-workflow', kind: 'behavior', panel: { n: 3, k: 2 }, source: source(completedAt),
trials: outcomes.map((outcome, i) => ({ trial: i + 1, outcome, ...(outcome === 'failed' ? { failure_class: 'timeout' as const } : {}) })),
});
test('every trial of a panel shares one identity; the panel policy is part of it', () => {
const key = (r: E2EShardReuseRequest) => { const x = e2eShardIdentity(r); if (x.status !== 'eligible') throw new Error(x.reason); return x.identity.key; };
const t1 = key(trialRequest(1));
expect(key(trialRequest(2, { env: laneEnv({ GSTACK_EVAL_TRIAL: '2' }) }))).toBe(t1);
expect(key(trialRequest(1, { panel: { ...panelPlan, quarantined: true } }))).not.toBe(t1);
expect(key(request())).not.toBe(t1);
});
test('a whole PASS panel receipt is reused per trial, a split PASS keeps its failed trial', () => {
const dir = path.join(scratch, 'panel-hit');
const env = laneEnv({ EVALS_CACHE_DIR: dir });
const reuse = prepareE2EShardReuse(trialRequest(2, { env }))!;
expect(reuse.lookupPanelTrial(2)).toBeNull();
writePanelReceipt(dir, panel(reuse.inputKey, ['passed', 'failed', 'passed']));
expect(reuse.lookupPanelTrial(2)).toMatchObject({ trial: { trial: 2, outcome: 'failed', failure_class: 'timeout' }, hit: { source: { runId: '1001/1' } } });
expect(reuse.lookupPanelTrial(1)!.trial.outcome).toBe('passed');
});
test('FAIL, partial, expired or negatively receipted panels are never reused', () => {
const dir = path.join(scratch, 'panel-miss');
const key = 'a'.repeat(64);
for (const receipt of [panel(key, ['passed', 'failed', 'failed']), panel(key, ['passed', 'passed']),
panel(key, ['passed', 'passed', 'passed'], Date.now() - 2 * 24 * 60 * 60 * 1000)]) {
writePanelReceipt(dir, receipt);
expect(readPanelReceipt(dir, key)).toBeNull();
}
writePanelReceipt(dir, panel(key, ['passed', 'passed', 'passed'], Date.now() - 5_000));
expect(readPanelReceipt(dir, key)).not.toBeNull();
writeNegativeReceipt(dir, { schema: 1, key, source: source(Date.now() - 1_000, '1002/1') });
expect(readPanelReceipt(dir, key)).toBeNull();
});
test('the planner ships one filtered set: a newer FAIL blocks an older PASS, an older FAIL does not', () => {
const from = path.join(scratch, 'select-from');
const to = path.join(scratch, 'select-to');
fs.mkdirSync(from, { recursive: true });
const [blockedKey, keptKey, panelKey] = ['1', '2', '3'].map(c => c.repeat(64));
const passReceipt = (key: string, completedAt: number) => fs.writeFileSync(path.join(from, `${key}.json`),
JSON.stringify({ schema: 1, proof: { source: source(completedAt) } }));
passReceipt(blockedKey, 1_000);
writeNegativeReceipt(from, { schema: 1, key: blockedKey, source: source(2_000, '1002/1') });
passReceipt(keptKey, 3_000);
writeNegativeReceipt(from, { schema: 1, key: keptKey, source: source(2_000, '1002/1') });
writePanelReceipt(from, panel(panelKey, ['passed', 'passed']));
const result = selectPlanReceipts(from, to);
expect(result.blocked.sort()).toEqual([`${blockedKey}.json`, `${panelKey}.panel.json`].sort());
expect(fs.readdirSync(to).sort()).toEqual([`${blockedKey}.fail.json`, `${keptKey}.fail.json`, `${keptKey}.json`].sort());
});
test('merging receipt stores keeps the newest file per name', () => {
const [a, b, out] = ['merge-a', 'merge-b', 'merge-out'].map(name => path.join(scratch, name));
const key = '4'.repeat(64);
writeNegativeReceipt(a, { schema: 1, key, source: source(5_000, '1/1') });
writeNegativeReceipt(b, { schema: 1, key, source: source(9_000, '2/1') });
expect(mergeReceiptDirs(out, [a, b, path.join(scratch, 'missing')])).toBe(2);
expect(JSON.parse(fs.readFileSync(path.join(out, `${key}.fail.json`), 'utf8')).source.runId).toBe('2/1');
expect(mergeReceiptDirs(out, [a])).toBe(0);
});
});
+11 -10
View File
@@ -1,18 +1,17 @@
import { expect, test } from 'bun:test';
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards';
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, expandTrialShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards';
import { FINDING_RETRY_BUDGETS, ALL_TIERS, SHARD_RESERVE_MS } from './helpers/eval-budgets';
import fs from 'node:fs';
import os from 'node:os';
import path from 'node:path';
for (const budget of FINDING_RETRY_BUDGETS) {
test(`${budget.file}: supervision preserves every existing attempt and retry`, () => {
test(`${budget.file}: supervision covers its one run of every case`, () => {
expect(budget.testMs).toBe(1_500_000);
// A 25-minute case is past RETRY_MAX_CASE_MS: a timed-out attempt is its verdict.
expect(budget.retries).toBe(0);
expect(retriesForFiles([budget.file])).toBe(budget.retries);
// Paid evals never retry: a timed-out case is its verdict.
expect(retriesForFiles([budget.file])).toBe(0);
expect(budget.shardReserveMs).toBe(SHARD_RESERVE_MS);
expect(budget.shardMs).toBe(budget.cases * budget.testMs * (budget.retries + 1) + budget.shardReserveMs);
expect(budget.shardMs).toBe(budget.cases * budget.testMs + budget.shardReserveMs);
expect(resolvePaidShardBudget([budget.file])).toEqual({ timeoutMs: budget.shardMs, source: 'registered', policyId: budget.id });
const source = fs.readFileSync(path.join(import.meta.dir, '..', budget.file), 'utf8');
if (budget.file === 'test/skill-e2e-plan-ceo-split-overflow.test.ts') {
@@ -176,18 +175,20 @@ test('single-slice manifest retains all registered files with one allocation', (
test('current detach supervision covers the live-census floor', () => {
const floorFor = (tier: 'gate' | 'periodic') => {
// Case-sharded files contribute one shard per case, exactly as the runner plans.
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier);
// Case-sharded files contribute one shard per case and isolated cases one
// shard per trial, exactly as the runner plans.
const files = expandTrialShards(expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier), tier).keys;
const excess = files.reduce((n, file) => n + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - DEFAULT_SHARD_TIMEOUT_MS), 0);
return Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05);
};
const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8'));
const periodicTimeout = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]);
const gateTimeout = Number(pkg.scripts['eval:bg:gate'].match(/--timeout\s+(\d+)/)[1]);
expect(floorFor('gate')).toBe(26_471);
expect(floorFor('gate')).toBe(21_725);
expect(gateTimeout).toBe(49_320);
expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate'));
expect(floorFor('periodic')).toBe(30_797);
expect(floorFor('periodic')).toBe(33_821);
expect(periodicTimeout).toBeGreaterThanOrEqual(floorFor('periodic'));
});
for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => {
+308 -3
View File
@@ -13,6 +13,13 @@ import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { aggregate, collectEvalFiles } from '../scripts/eval-flake-rank';
import { manualReviewFixture } from './helpers/manual-judge-review-fixture';
import {
analyzePassRates, attributeLegacyRecord, backfillEvalFiles, caseSeriesIdentities, downloadRunArtifacts, fisherOneSidedLower,
formatPassRates, holmRejections, listWeeklyRuns, quarantinePolicyProblems, quarantineRunsSince, readTrialOutcomeDir,
wilsonInterval, type HistoryFetcher, type PassRatePolicy, type QuarantineEntry, type Registry, type TrialRecord,
} from '../scripts/eval-flake-rank';
import { EVAL_POLICY } from './helpers/periodic-exclude-data';
import { TRIAL_OUTCOME_SCHEMA, formatTrialOutcomes } from './helpers/eval-store';
const entry = (name: string, passed: boolean, attempt: number) => ({
name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1,
@@ -39,9 +46,10 @@ describe('eval-flake-rank aggregate', () => {
const display = spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), '--dir', dir],
{ encoding: 'utf8', timeout: 10_000 });
expect(display.status, display.stderr).toBe(0);
expect(display.stdout).toContain('fails/runs manual');
expect(display.stdout).toContain('0/1');
expect(display.stdout).toContain(manual.name);
// pass-rates view: the prior automated pass is the one scored pre-policy
// trial; the manual acceptance is counted in its own column, never scored.
expect(display.stdout).toContain('pre-policy manual case');
expect(display.stdout).toMatch(new RegExp(`1/1 \\[[^\\]]+\\]\\s+1 ${manual.name.replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')}`));
fs.writeFileSync(path.join(dir, 'invalid-retry.json'), run([
{ ...ordinary, attempt: 1 }, { ...manual, attempt: 2 },
]));
@@ -97,3 +105,300 @@ describe('eval-flake-rank aggregate', () => {
fs.rmSync(dir, { recursive: true, force: true });
});
});
// --- pass-rates ---
const registry: Registry = {
kinds: { 'rule-a': 'rule', 'beh-b': 'behavior', 'gate-c': 'rule', 'mar-d': 'rule', 'judge one': 'judge',
...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, 'rule'])) },
tiers: { 'rule-a': 'periodic', 'beh-b': 'periodic', 'gate-c': 'gate', 'mar-d': 'marathon',
...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, i < 9 ? 'gate' : 'periodic'])) },
touchfiles: { 'rule-a': ['test/skill-e2e-a.test.ts', 'a/**'], 'beh-b': ['test/skill-e2e-b.test.ts', 'b/**'],
'gate-c': ['test/skill-e2e-shared.test.ts'], 'mar-d': ['test/skill-e2e-shared.test.ts'] },
judgeTouchfiles: { 'judge one': ['j/SKILL.md'] },
globals: ['harness/**'],
testNames: { 'gate-c': '/gate c labeled' },
};
let clock = 0;
function trial(id: string, outcome: 'passed' | 'failed' | 'skipped', extra: Partial<TrialRecord> = {}): TrialRecord {
clock += 1;
return {
schema: TRIAL_OUTCOME_SCHEMA, case: id, file: 'test/x.test.ts', tier: registry.tiers[id] ?? 'judge',
kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1, outcome,
...(outcome === 'failed' ? { failure_class: 'assertion' as const } : {}),
duration_ms: 1, cost_usd: 0, model: 'model-x', cli_version: '2.1.284', policy_version: 1, quarantined: false,
execution: 'executed', source: 'shard', run_id: `run-${clock}`, recorded_at: new Date(Date.UTC(2026, 9, 1) + clock * 60_000).toISOString(),
series_identity: 'id-1', ...extra,
};
}
const many = (id: string, passes: number, fails: number, extra: Partial<TrialRecord> = {}) =>
[...Array.from({ length: passes }, () => trial(id, 'passed', extra)), ...Array.from({ length: fails }, () => trial(id, 'failed', extra))];
const analyze = (records: TrialRecord[], quarantine: Record<string, QuarantineEntry> = {}, extra = {}) =>
analyzePassRates(records, { registry, quarantine, now: Date.UTC(2026, 9, 2), ...extra });
const qEntry = (overrides: Partial<QuarantineEntry> = {}): QuarantineEntry => ({
reason: 'Detector graded the posture wording; 7 of 10 fresh trials failed only the regex, transcripts attached.',
failureClass: 'detector', tracking: 'TODOS.md "x"', owner: 'garrytan', enteredAt: '2026-09-29',
exit: '>= 97% over >= 10 trials on the current identity', ...overrides,
});
describe('pass-rates statistics', () => {
test('Wilson bounds match the documented policy arithmetic', () => {
expect(wilsonInterval(10, 10).lo).toBeCloseTo(0.7225, 4);
expect(wilsonInterval(6, 6).lo).toBeCloseTo(0.6097, 4);
expect(wilsonInterval(125, 125).lo).toBeCloseTo(0.9702, 4);
expect(wilsonInterval(10, 10).hi).toBe(1);
expect(wilsonInterval(0, 0)).toEqual({ lo: 0, hi: 1 });
const mid = wilsonInterval(7, 10);
expect(mid.lo).toBeGreaterThan(0.39); expect(mid.hi).toBeLessThan(0.9);
});
test('one-sided Fisher exact matches a known table and is one-sided', () => {
expect(fisherOneSidedLower(4, 6, 6, 6)).toBeCloseTo(0.227272727, 8);
expect(fisherOneSidedLower(6, 6, 4, 6)).toBe(1);
expect(fisherOneSidedLower(0, 10, 10, 10)).toBeLessThan(1e-4);
});
test('Holm rejects step-down and stops at the first non-rejection', () => {
expect([...holmRejections([0.001, 0.02, 0.04], 0.05)].sort()).toEqual([0, 1, 2]);
expect([...holmRejections([0.001, 0.03, 0.04], 0.05)]).toEqual([0]);
expect([...holmRejections([0.03, 0.04], 0.05)]).toEqual([]);
expect([...holmRejections([], 0.05)]).toEqual([]);
});
});
describe('pass-rates labels', () => {
test('thin history is INCONCLUSIVE, and after this PR every series starts there', () => {
const report = analyze(many('rule-a', 9, 0));
expect(report.cases[0]).toMatchObject({ case: 'rule-a', label: 'INCONCLUSIVE' });
expect(formatPassRates(analyze([]))).toContain('every series starts INCONCLUSIVE');
});
test('PASSING, FLAKY and FAILING come from the interval against the entry rate', () => {
expect(analyze(many('rule-a', 12, 0)).cases[0]!.label).toBe('PASSING');
expect(analyze(many('beh-b', 10, 1)).cases[0]!.label).toBe('FLAKY');
expect(analyze(many('beh-b', 2, 10)).cases[0]!.label).toBe('FAILING');
});
test('BROKEN: the latest run is 0/n after a prior interval at or above the entry rate', () => {
const prior = many('beh-b', 80, 0, { run_id: 'old' });
const latest = [1, 2, 3].map(n => trial('beh-b', 'failed', { run_id: 'new', trial: n, panel: { n: 3, k: 2 } }));
expect(analyze([...prior, ...latest]).cases[0]!.label).toBe('BROKEN');
});
test('skipped trials carry no verdict; infra failures count as failed trials', () => {
const stats = analyze([...many('rule-a', 10, 0), trial('rule-a', 'skipped'),
trial('rule-a', 'failed', { failure_class: 'infra' })]).cases[0]!.current!;
expect(stats).toMatchObject({ passes: 10, trials: 11, infra: 1 });
});
test('a new identity, model or CLI starts a new series; earlier series stay visible', () => {
const report = analyze([...many('rule-a', 10, 0), ...many('rule-a', 3, 0, { series_identity: 'id-2' }),
...many('rule-a', 2, 0, { series_identity: 'id-2', cli_version: '2.1.285' })]);
const c = report.cases[0]!;
expect(c.series).toHaveLength(3);
expect(c.current).toMatchObject({ identity: 'id-2', cli: '2.1.285', trials: 2 });
expect(c.previous).toMatchObject({ identity: 'id-2', cli: '2.1.284', trials: 3 });
expect(c.label).toBe('INCONCLUSIVE');
});
});
describe('pass-rates alarms count post-policy trials of the current series only', () => {
test('backfilled pre-policy failures are displayed but never alarm', () => {
const report = analyze(many('rule-a', 2, 20, { policy_version: 0, source: 'backfill' }));
expect(report.alarms).toEqual([]);
expect(report.cases[0]!.prePolicy).toMatchObject({ passes: 2, trials: 22 });
expect(report.cases[0]!.label).toBe('INCONCLUSIVE');
});
test('drift proposes quarantine for a blocking case below the entry rule; a rule case is flagged as behaving like behavior', () => {
const kinds = analyze([...many('rule-a', 8, 2), ...many('beh-b', 8, 2), ...many('mar-d', 0, 10)]).alarms.map(a => `${a.kind}:${a.case}`);
expect([...kinds].sort()).toEqual(['drift:beh-b', 'drift:rule-a', 'rule-as-behavior:mar-d', 'rule-as-behavior:rule-a']);
expect(analyze(many('rule-a', 19, 1)).alarms).toEqual([]);
});
test('the Fisher regression alarm needs the minimum trials on both sides', () => {
const old = many('gate-c', 6, 0, { series_identity: 'old' });
const fresh = many('gate-c', 0, 6, { series_identity: 'new' });
expect(analyze([...old, ...fresh]).alarms.map(a => a.kind)).toContain('regression');
expect(analyze([...old, ...fresh.slice(0, 5)]).alarms.map(a => a.kind)).not.toContain('regression');
});
test('quarantine exit, expiry and cap', () => {
const exit = analyze(many('beh-b', 10, 0), { 'beh-b': qEntry() }).alarms.map(a => a.kind);
expect(exit).toContain('quarantine-exit');
expect(exit).not.toContain('drift');
const weekly = Array.from({ length: 8 }, (_, i) => new Date(Date.UTC(2026, 8, 30) + i * 7 * 86_400_000).toISOString());
expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly }).alarms.map(a => a.kind)).toContain('quarantine-expired');
expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly.slice(0, 7) }).alarms.map(a => a.kind)).not.toContain('quarantine-expired');
expect(quarantineRunsSince('2026-09-01', undefined, Date.UTC(2026, 9, 27))).toBe(8);
expect(quarantineRunsSince('not a date', undefined, 0)).toBe(Number.POSITIVE_INFINITY);
});
});
describe('quarantine policy', () => {
const policy: PassRatePolicy = EVAL_POLICY;
const now = Date.UTC(2026, 9, 2);
test('a valid entry has no problems', () => {
expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]);
});
test('a product defect, a missing diagnosis or field, a bad date or a non-blocking case is rejected', () => {
const problems = (quarantine: Record<string, QuarantineEntry>) => quarantinePolicyProblems(quarantine, registry, policy, now).map(p => p.message);
expect(problems({ 'beh-b': qEntry({ failureClass: 'product' as QuarantineEntry['failureClass'] }) }).join()).toContain('never quarantined');
expect(problems({ 'beh-b': qEntry({ reason: 'flaky' }) }).join()).toContain('written diagnosis');
expect(problems({ 'beh-b': qEntry({ owner: ' ' }) }).join()).toContain('missing owner');
expect(problems({ 'beh-b': qEntry({ enteredAt: '09/29/2026' }) }).join()).toContain('YYYY-MM-DD');
expect(problems({ 'beh-b': qEntry({ enteredAt: '2027-01-01' }) }).join()).toContain('future');
expect(problems({ 'mar-d': qEntry() }).join()).toContain('not blocking');
expect(problems({ 'judge one': qEntry() }).join()).toContain('no registered E2E case');
expect(problems({ ghost: qEntry() }).join()).toContain('no registered E2E case');
});
test('at most 10% of a tier may be quarantined', () => {
// 11 periodic cases in the fixture registry: the cap is 1.
expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]);
const over = quarantinePolicyProblems({ 'beh-b': qEntry(), 'rule-a': qEntry() }, registry, policy, now);
expect(over.map(p => p.kind)).toEqual(['quarantine-cap']);
expect(over[0]!.message).toContain('2 quarantined periodic cases exceed the 10% cap (1 of 11)');
});
});
describe('pass-rates inputs', () => {
test('trial-outcomes JSONL is schema-validated; invalid lines are reported, never guessed', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-'));
const valid = trial('rule-a', 'passed');
fs.mkdirSync(path.join(dir, 'nested'));
fs.writeFileSync(path.join(dir, 'nested', 'trial-outcomes.jsonl'), formatTrialOutcomes([valid]) + '{"schema":"other"}\nnot json\n');
fs.writeFileSync(path.join(dir, 'unrelated.jsonl'), formatTrialOutcomes([valid]));
const read = readTrialOutcomeDir(dir);
expect(read.records).toHaveLength(1);
expect(read.records[0]).toMatchObject({ case: 'rule-a', series_identity: 'id-1' });
expect(read.errors).toHaveLength(2);
fs.rmSync(dir, { recursive: true, force: true });
});
test('legacy records attribute by shard suffix, id, label or single-owner file, else stay unattributed', () => {
expect(attributeLegacyRecord('/anything', 'skill-e2e-b--beh-b', registry)).toBe('beh-b');
expect(attributeLegacyRecord('rule-a', 'skill-e2e-zzz', registry)).toBe('rule-a');
expect(attributeLegacyRecord('/Rule a', 'skill-e2e-zzz', registry)).toBe('rule-a');
expect(attributeLegacyRecord('/rule a extra', 'skill-e2e-zzz', registry)).toBeNull();
expect(attributeLegacyRecord('/gate c labeled', undefined, registry)).toBe('gate-c');
expect(attributeLegacyRecord('/a display name', 'skill-e2e-a', registry)).toBe('rule-a');
expect(attributeLegacyRecord('/shared display', 'skill-e2e-shared', registry)).toBeNull();
});
test('backfill keeps only first attempts, defaults a missing attempt to 1, and labels records pre-policy', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-backfill-'));
fs.writeFileSync(path.join(dir, 'run.json'), run([
{ ...entry_('rule-a', false, 1), exit_reason: 'timeout' }, entry_('rule-a', true, 2),
{ name: 'beh-b', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 },
entry_('/unknown display', true, 1),
], { shard: 'skill-e2e-zzz' }));
const { records, unattributed } = backfillEvalFiles(collectEvalFiles(dir), { run_id: '42', sha: 'abc' }, registry);
expect(records.map(r => [r.case, r.outcome, r.failure_class, r.policy_version, r.source, r.run_id]))
.toEqual([['rule-a', 'failed', 'timeout', 0, 'backfill', '42'], ['beh-b', 'passed', undefined, 0, 'backfill', '42']]);
expect(formatTrialOutcomes(records)).toContain(TRIAL_OUTCOME_SCHEMA);
expect(unattributed).toEqual(['/unknown display']);
fs.rmSync(dir, { recursive: true, force: true });
});
test('series identity follows the case\'s own touchfiles, not GLOBAL_TOUCHFILES', () => {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-series-'));
const git = (...args: string[]) => spawnSync('git', args, { cwd: root, encoding: 'utf8', timeout: 10_000 });
for (const [file, body] of [['a/x.ts', '1'], ['b/y.ts', '1'], ['harness/run.ts', '1'], ['test/skill-e2e-a.test.ts', '1']] as const) {
fs.mkdirSync(path.join(root, path.dirname(file)), { recursive: true });
fs.writeFileSync(path.join(root, file), body);
}
const snapshot = () => { expect(git('add', '-A').status).toBe(0); return caseSeriesIdentities(['rule-a', 'beh-b'], root, registry); };
expect(git('init', '-q').status).toBe(0);
const first = snapshot();
expect(first['rule-a']).not.toBe(first['beh-b']);
fs.writeFileSync(path.join(root, 'harness/run.ts'), '2');
expect(snapshot()).toEqual(first);
fs.writeFileSync(path.join(root, 'a/x.ts'), '2');
const next = snapshot();
expect(next['rule-a']).not.toBe(first['rule-a']);
expect(next['beh-b']).toBe(first['beh-b']);
fs.rmSync(root, { recursive: true, force: true });
});
});
describe('pass-rates history fetch (injected, no network)', () => {
function storedZip(files: Record<string, string>): Buffer {
const locals: Buffer[] = [], centrals: Buffer[] = [];
let offset = 0;
for (const [name, text] of Object.entries(files)) {
const data = Buffer.from(text), fileName = Buffer.from(name), crc = Bun.hash.crc32(data) >>> 0;
const local = Buffer.alloc(30); local.writeUInt32LE(0x04034b50, 0); local.writeUInt16LE(20, 4);
local.writeUInt32LE(crc, 14); local.writeUInt32LE(data.length, 18); local.writeUInt32LE(data.length, 22); local.writeUInt16LE(fileName.length, 26);
const central = Buffer.alloc(46); central.writeUInt32LE(0x02014b50, 0); central.writeUInt16LE(20, 4); central.writeUInt16LE(20, 6);
central.writeUInt32LE(crc, 16); central.writeUInt32LE(data.length, 20); central.writeUInt32LE(data.length, 24);
central.writeUInt16LE(fileName.length, 28); central.writeUInt32LE(offset, 42);
locals.push(local, fileName, data); centrals.push(central, fileName);
offset += 30 + fileName.length + data.length;
}
const size = centrals.reduce((sum, b) => sum + b.length, 0);
const end = Buffer.alloc(22); end.writeUInt32LE(0x06054b50, 0); end.writeUInt16LE(Object.keys(files).length, 8);
end.writeUInt16LE(Object.keys(files).length, 10); end.writeUInt32LE(size, 12); end.writeUInt32LE(offset, 16);
return Buffer.concat([...locals, ...centrals, end]);
}
test('lists runs per branch, deduplicated and newest first', () => {
const fetcher: HistoryFetcher = {
listRuns: (_repo, _workflow, branch) => branch === 'main'
? [{ id: 1, attempt: 1, sha: 'a', branch, createdAt: '2026-09-01T00:00:00Z' }, { id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }]
: [{ id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }, { id: 2, attempt: 2, sha: 'b', branch, createdAt: '2026-09-08T00:00:00Z' }],
listArtifacts: () => [], downloadZip: () => { throw new Error('unused'); },
};
expect(listWeeklyRuns({ repo: 'o/r', workflow: 'evals-periodic.yml', branches: ['feature', 'main'], limit: 10, fetcher }).map(r => r.id)).toEqual([3, 2, 1]);
});
test('downloads only matching, bounded artifacts once, and caches them', () => {
const cacheDir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cache-'));
const downloads: number[] = [];
const fetcher: HistoryFetcher = {
listRuns: () => [],
listArtifacts: () => [{ id: 10, name: 'trial-outcomes-gate', size: 100 }, { id: 11, name: 'paid-slice-1', size: 100 },
{ id: 12, name: 'trial-outcomes-huge', size: 10 ** 9 }, { id: 13, name: 'trial-outcomes/../escape', size: 1 }],
downloadZip: (_repo, id, destination) => { downloads.push(id); fs.writeFileSync(destination, storedZip({ 'trial-outcomes.jsonl': formatTrialOutcomes([trial('rule-a', 'passed')]) })); },
};
const options = { repo: 'o/r', run: { id: 7, attempt: 1, sha: 's', branch: 'main', createdAt: '' }, cacheDir, fetcher,
match: (name: string) => name.startsWith('trial-outcomes') };
const dirs = downloadRunArtifacts(options);
expect(downloads).toEqual([10]);
expect(dirs).toHaveLength(1);
expect(readTrialOutcomeDir(dirs[0]!).records.map(r => r.case)).toEqual(['rule-a']);
expect(downloadRunArtifacts(options)).toEqual(dirs);
expect(downloads).toEqual([10]);
fs.rmSync(cacheDir, { recursive: true, force: true });
});
});
describe('pass-rates CLI', () => {
const cli = (args: string[]) => spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), ...args],
{ encoding: 'utf8', timeout: 20_000 });
test('--dir prints per-case pass rates; --gate fails only on ACTION REQUIRED', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cli-'));
const id = 'plan-ceo-review-format-mode';
const records = Array.from({ length: 12 }, (_, i) => ({ ...trial(id, i < 11 ? 'passed' : 'failed'), kind: 'behavior' as const, tier: 'periodic' }));
fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records));
const shown = cli(['--dir', dir, '--case', id]);
expect(shown.status, shown.stderr).toBe(0);
expect(shown.stdout).toContain(`11/12 [`);
expect(shown.stdout).toMatch(new RegExp(`FLAKY\\s+behavior\\s+periodic.*${id}`));
expect(shown.stdout).toContain('ACTION REQUIRED');
expect(shown.stdout).toContain(`[drift] ${id} passes 11/12`);
expect(cli(['--dir', dir, '--gate']).status).toBe(1);
fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records.slice(0, 11)));
const clean = cli(['--dir', dir, '--gate', '--json']);
expect(clean.status, clean.stdout).toBe(0);
expect(JSON.parse(clean.stdout).cases[0]).toMatchObject({ case: id, label: 'PASSING', current: { passes: 11, trials: 11 } });
fs.rmSync(dir, { recursive: true, force: true });
});
});
function entry_(name: string, passed: boolean, attempt: number) {
return { name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1 };
}
+107
View File
@@ -0,0 +1,107 @@
/**
* Eval kind registry (E2E_KINDS / BEHAVIOR_WHY in touchfiles-data.ts). The
* kind fixes a case's trial policy before the run, so the registry must cover
* every live case exactly once, every behavior case must name its tolerated
* deviation, and a behavior case must be isolatable as its own trial shard.
* A kind edit must re-select the case in the PR lane (map-diff).
*/
import { describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { BEHAVIOR_WHY, E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data';
import { diffTouchfileMapsCore, type TouchfileMaps } from './helpers/test-selection';
import { CASE_TEST_NAMES, fileCaseRegistration } from '../scripts/test-paid-shards';
import { isPaidTestFile } from './helpers/paid-test-set';
const ROOT = path.resolve(import.meta.dir, '..');
const KIND_RULE = "Pick the kind by what can make the verdict differ between two runs of the same commit: 'rule' when nothing "
+ "stochastic decides it or it checks a contract the product must meet every run (the default); 'behavior' when a live "
+ "model choice decides it and a sub-100% per-trial rate is acceptable (add a BEHAVIOR_WHY line); 'judge' when the only "
+ 'stochastic step is an LLM judge scoring a fixed input.';
const liveIds = [...Object.keys(E2E_TIERS), ...Object.keys(LLM_JUDGE_TOUCHFILES)];
const behaviorIds = Object.keys(E2E_KINDS).filter(id => E2E_KINDS[id] === 'behavior').sort();
describe('E2E_KINDS registry', () => {
test('every live case has exactly one kind and no kind names a dead case', () => {
const missing = liveIds.filter(id => !(id in E2E_KINDS));
expect(missing.length, missing.length ? `add to E2E_KINDS:\n${missing.map(id => ` '${id}': 'rule', // <reason>`).join('\n')}\n${KIND_RULE}` : '').toBe(0);
const unknown = Object.keys(E2E_KINDS).filter(id => !liveIds.includes(id));
expect(unknown, `E2E_KINDS names ids that are neither E2E_TIERS nor LLM_JUDGE_TOUCHFILES keys`).toEqual([]);
expect(new Set(liveIds).size).toBe(liveIds.length);
});
test('kinds are rule, behavior or judge; every LLM-judge entry is judge-kind', () => {
for (const [id, kind] of Object.entries(E2E_KINDS)) expect(['rule', 'behavior', 'judge'], id).toContain(kind);
for (const id of Object.keys(LLM_JUDGE_TOUCHFILES)) expect(E2E_KINDS[id], `${id}: a workflow judge scores a fixed input`).toBe('judge');
});
test('BEHAVIOR_WHY names the tolerance of exactly the behavior cases', () => {
expect(Object.keys(BEHAVIOR_WHY).sort()).toEqual(behaviorIds);
for (const id of behaviorIds) {
expect(BEHAVIOR_WHY[id]!.trim().length, `${id}: BEHAVIOR_WHY must say why an occasional deviation is acceptable`).toBeGreaterThanOrEqual(30);
}
});
test('a behavior case is an isolatable trial shard: known literal registration and an exact Bun test name', () => {
for (const id of behaviorIds) {
const files = E2E_TOUCHFILES[id]!.filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file));
expect(files.length, `${id}: no paid test file registers it`).toBeGreaterThan(0);
for (const file of files) {
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
expect(fileCaseRegistration(file, source).known, `${id}: ${file} has a computed registration; behavior needs a literal one`).toBe(true);
const name = CASE_TEST_NAMES[id] ?? id;
const literal = new RegExp(`\\b(?:test(?:\\.serial|\\.concurrent)?|testIfSelected|testConcurrentIfSelected)\\(\\s*(['"\`])${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\1`);
expect(literal.test(source), `${id}: ${file} must register the Bun test named '${name}'`).toBe(true);
}
}
});
test('the classification is the reviewed one: rule by default, 22 behavior, 25 judge', () => {
const counts = Object.values(E2E_KINDS).reduce<Record<string, number>>((acc, kind) => ({ ...acc, [kind]: (acc[kind] ?? 0) + 1 }), {});
expect(counts).toEqual({ rule: liveIds.length - 22 - 25, behavior: 22, judge: 25 });
// Contract-shaped cases stay rule: ask-before-decide, plan-mode no-writes,
// mandated steps, secrets, and the batching floor never ride a majority.
for (const id of ['plan-ceo-mode-routing', 'plan-eng-multi-finding-batching', 'plan-design-review-plan-mode',
'plan-eng-review-plan-mode', 'plan-ceo-section-loading', 'setup-gbrain-bad-token', 'qa-only-no-fix', 'review-sql-injection']) {
expect(E2E_KINDS[id], id).toBe('rule');
}
});
});
describe('kind edits re-select their case (map-diff)', () => {
const base = (): TouchfileMaps => ({
E2E_TOUCHFILES: { alpha: ['a/**'], beta: ['b/**'] },
E2E_TIERS: { alpha: 'gate', beta: 'periodic' },
LLM_JUDGE_TOUCHFILES: { 'judge one': ['j/SKILL.md'] },
GLOBAL_TOUCHFILES: [],
E2E_KINDS: { alpha: 'rule', beta: 'rule', 'judge one': 'judge' },
BEHAVIOR_WHY: {},
});
test('a rule -> behavior flip selects exactly that case', () => {
const next = base();
next.E2E_KINDS = { ...next.E2E_KINDS, beta: 'behavior' };
next.BEHAVIOR_WHY = { beta: 'tolerated deviation' };
expect(diffTouchfileMapsCore(base(), next).changedTests).toEqual(['beta']);
});
test('a BEHAVIOR_WHY edit alone selects its case', () => {
const old = base(); old.E2E_KINDS!.beta = 'behavior'; old.BEHAVIOR_WHY = { beta: 'one' };
const next = base(); next.E2E_KINDS!.beta = 'behavior'; next.BEHAVIOR_WHY = { beta: 'two' };
expect(diffTouchfileMapsCore(old, next).changedTests).toEqual(['beta']);
});
test('a base revision without the kind maps selects every key', () => {
const old = base(); delete old.E2E_KINDS; delete old.BEHAVIOR_WHY;
expect(diffTouchfileMapsCore(old, base()).changedTests).toEqual(['alpha', 'beta', 'judge one']);
});
test('dropping a kind entry while the case lives on counts as changed, not removed', () => {
const next = base(); delete next.E2E_KINDS!.alpha;
const result = diffTouchfileMapsCore(base(), next);
expect(result.changedTests).toEqual(['alpha']);
expect(result.removedTests).toEqual([]);
});
});
+65
View File
@@ -263,3 +263,68 @@ describe('shared setup composites (every paid lane)', () => {
}
});
});
describe('panel verdict surfaces (eval reliability policy)', () => {
type AnyJob = { if?: string; needs?: string[]; permissions?: Record<string, string>; outputs?: Record<string, string>;
strategy?: { 'max-parallel': number }; steps: Array<Step & { if?: string; uses?: string }> };
const jobsOf = (source: string) => (Bun.YAML.parse(source) as { jobs: Record<string, AnyJob> }).jobs;
test('planners size the capacity preflight with their executor cap', () => {
for (const [source, executor, manifest] of [[evalsYml, 'eval-slices', '/tmp/paid-plan/manifest.json'],
[periodicYml, 'eval-slices', '/tmp/paid-plan/manifest.json'], [periodicYml, 'gate-census', '/tmp/gate-census-plan/manifest.json']] as const) {
const jobs = jobsOf(source);
const emit = jobs['plan-slices']!.steps.find(step => step.run?.includes(`--emit-plan ${manifest} `))!;
const cap = Number(/--max-parallel (\d+)/.exec(emit.run!)?.[1]);
expect(cap, `${executor}: --max-parallel`).toBe(jobs[executor]!.strategy!['max-parallel']);
}
});
test('slice artifacts are attempt-scoped and never merged into one tree', () => {
for (const source of [evalsYml, periodicYml, marathonYml]) {
const jobs = jobsOf(source);
const uploads = Object.values(jobs).flatMap(job => job.steps).filter(step => step.uses?.startsWith('actions/upload-artifact@'))
.map(step => step.with?.name ?? '').filter(name => /slice|census-\$/.test(name));
expect(uploads.length).toBeGreaterThan(0);
for (const name of uploads) expect(name, name).toContain('-a${{ github.run_attempt }}');
const downloads = Object.values(jobs).flatMap(job => job.steps).filter(step => step.uses?.startsWith('actions/download-artifact@') && step.with?.pattern);
for (const step of downloads) expect((step.with as Record<string, unknown>)['merge-multiple'], step.with!.pattern).toBeUndefined();
}
});
test('the PR comment reads collector-outcomes v2 and never recomputes a verdict', () => {
const comment = evalsYml.slice(evalsYml.indexOf(' slices-comment:'));
expect(comment).toContain('.version == 2');
expect(comment).toContain("jq -r '.failures[]'");
expect(comment).toContain('name: report-verdict-a${{ github.run_attempt }}');
expect(evalsYml).not.toContain('group_by(.name)');
expect(comment).not.toMatch(/paid-slice-/);
const report = jobsOf(evalsYml)['slices-report']!;
expect(report.steps.some(step => step.run?.includes('scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl'))).toBe(true);
expect(report.steps.some(step => step.with?.name?.startsWith('trial-outcomes-'))).toBe(true);
});
test('the weekly report gates on pass-rate history, closes its issue on green, and re-dispatches INFRA-only reds once', () => {
const jobs = jobsOf(periodicYml);
const report = jobs.report!;
expect(report.permissions).toEqual({ contents: 'read', issues: 'write', actions: 'read' });
const gate = report.steps.find(step => step.id === 'pass-rates')!;
expect(gate.run).toContain('bun run eval:pass-rates --gate --runs 10');
expect(gate.if).toBe('always()');
for (const name of ['Upsert tracking issue on failure', 'Fail the workflow when reconciliation failed']) {
expect(report.steps.find(step => step.name === name)!.if).toContain("steps.pass-rates.outputs.exit != '0'");
}
const upsert = report.steps.find(step => step.name === 'Upsert tracking issue on failure')!;
expect(upsert.run).toContain('report-summary.md');
expect(report.steps.find(step => step.name === 'Close the tracking issue on a green run')!.run).toContain('gh issue close');
expect(report.steps.filter(step => step.with?.name?.startsWith('trial-outcomes-')).length).toBe(2);
const redispatch = jobs.redispatch!;
expect([redispatch.needs].flat()).toEqual(['report']);
expect(redispatch.permissions).toEqual({ actions: 'write' });
expect(redispatch.if).toBe("${{ !cancelled() && needs.report.outputs.redispatch == 'true' }}");
expect(redispatch.steps[0]!.run).toContain('-f redispatch_of="$GITHUB_RUN_ID"');
const classify = report.steps.find(step => step.id === 'verdict')!;
expect(classify.run).toContain('.verdict.redispatchEligible == true');
expect(classify.run).toContain('[ -z "$REDISPATCH_OF" ]');
expect(periodicYml).toMatch(/group: evals-periodic\$\{\{ inputs\.redispatch_of/);
});
});
+48 -100
View File
@@ -47,132 +47,80 @@ export const ALL_TIERS = {
export const SHARD_RESERVE_MS = 2 * 60_000;
/**
* Retry policy (approved 2026-09-29): a timed-out attempt is a verdict. Bun's
* --retry reruns a failed case after it may have spent its whole budget, so an
* automatic retry is kept only where one more attempt is short: every case of
* the file has a per-attempt budget of at most RETRY_MAX_CASE_MS, the CAPTURE
* tier plus its recording grace. Those failures are fast flake classes (API
* blips, tool hiccups) and a retry costs at most one more short attempt. Files
* with any longer case run once. Per-case budgets never change with this rule.
* Retry policy (approved 2026-09-29, eval reliability wave): paid evals never
* retry. Each case's kind (E2E_KINDS) fixes its trials before the run: `rule`
* one trial, `behavior` a panel of EVAL_POLICY.panel independent trials, and
* `judge` one case that samples its judge panel internally. A failed verdict
* is final for that run; a manual re-run adds trials under a new run attempt
* and never replaces the original verdict. Rows below keep only wall
* supervision; per-case budgets never change with this rule.
*/
export const RETRY_MAX_CASE_MS = CAPTURE_MS + 15_000;
export function retriesWithinCaseCap(caseMs: number, configuredRetries: number): number {
return caseMs <= RETRY_MAX_CASE_MS ? configuredRetries : 0;
}
/**
* Unregistered paid files that keep one automatic retry: every case budget is
* JUDGE or CAPTURE tier (test/paid-retry-supervision.test.ts scans each source).
* Registered rows below derive retries from their declared caseMs; every other
* paid file runs once.
*/
export const SHORT_CASE_RETRY_FILES: readonly string[] = [
'test/codex-e2e-sol-scope.test.ts',
'test/llm-judge-recommendation.test.ts',
'test/skill-e2e-ask-user-question-format-compliance.test.ts',
'test/skill-e2e-benchmark-providers.test.ts',
'test/skill-e2e-bws.test.ts',
'test/skill-e2e-context-skills.test.ts',
'test/skill-e2e-coverage-audit.test.ts',
'test/skill-e2e-diagram.test.ts',
'test/skill-e2e-first-task-scaffold.test.ts',
'test/skill-e2e-gbrain-roundtrip-local.test.ts',
'test/skill-e2e-hermetic-canary.test.ts',
'test/skill-e2e-investigate-owned-completion.test.ts',
'test/skill-e2e-investigate-owned-termination.test.ts',
'test/skill-e2e-learnings.test.ts',
'test/skill-e2e-plan-tune.test.ts',
'test/skill-e2e-qa-functional-fix.test.ts',
'test/skill-e2e-qa-functional.test.ts',
'test/skill-e2e-review-army.test.ts',
'test/skill-e2e-review.test.ts',
'test/skill-e2e-session-intelligence.test.ts',
'test/skill-e2e-setup-gbrain-bad-token.test.ts',
'test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts',
'test/skill-e2e-setup-gbrain-remote.test.ts',
'test/skill-e2e-ship-hook-consent.test.ts',
'test/skill-e2e-ship-hook-refresh.test.ts',
'test/skill-e2e-ship-skip.test.ts',
'test/skill-e2e-sync-gbrain-readiness.test.ts',
'test/skill-e2e-third-party-actions.test.ts',
'test/skill-e2e-triage.test.ts',
'test/skill-routing-e2e.test.ts',
];
/** Whole-file supervision covers every attempt the retry policy allows.
* These fixtures allow 25 minutes per case, so they run once.
/** Whole-file supervision for one run of every case.
* These fixtures allow 25 minutes per case.
* Reserve the sequential upper bound even when Bun runs sibling cases together.
*/
export const FINDING_RETRY_BUDGETS = [
{ file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 },
].map(({ file, cases }) => {
const retries = retriesWithinCaseCap(1_500_000, 1);
return {
file, cases,
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
testMs: 1_500_000,
caseMs: 1_500_000,
retries,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: cases * 1_500_000 * (retries + 1) + SHARD_RESERVE_MS,
};
});
].map(({ file, cases }) => ({
file, cases,
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
testMs: 1_500_000,
caseMs: 1_500_000,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: cases * 1_500_000 + SHARD_RESERVE_MS,
}));
/** Three existing captures in one 16-minute case, so the file runs once. */
/** Three existing captures in one 16-minute case. */
export const AUQ_CONSISTENCY_RETRY_BUDGET = {
file: 'test/skill-e2e-auq-consistency.test.ts',
id: 'auq-consistency-existing-retry-v1',
cases: 1,
testMs: 3 * CAPTURE_MS + 60_000,
caseMs: 3 * CAPTURE_MS + 60_000,
retries: retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1),
shardReserveMs: SHARD_RESERVE_MS,
shardMs: (3 * CAPTURE_MS + 60_000) * (retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1) + 1) + SHARD_RESERVE_MS,
shardMs: 3 * CAPTURE_MS + 60_000 + SHARD_RESERVE_MS,
} as const;
/** These fixtures have a fixed case count in every supported tier. */
export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET];
/** Whole-file walls cover all existing cases and every allowed attempt, even if
* Bun runs them sequentially. Mixed-tier files reserve their larger complete
* tier, never a currently selected subset. caseMs is the longest single case
* budget, which decides the retry (RETRY_MAX_CASE_MS). These rows add no
* case-count or model-work policy. The 10-second terms preserve the existing
* Codex/recording finalization grace.
/** Whole-file walls cover all existing cases, even if Bun runs them
* sequentially. Mixed-tier files reserve their larger complete tier, never a
* currently selected subset. caseMs is the longest single case budget, the
* wall of one isolated case shard. These rows add no case-count or model-work
* policy. The 10-second terms preserve the existing Codex/recording
* finalization grace.
*/
export const FILE_RETRY_BUDGETS = [
...STRICT_RETRY_CASE_BUDGETS,
...[
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000 },
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS },
// Seventeen workflow judges include their 10s recording grace; the other
// seven judges retain 120s. Supervise all 24 and the existing one retry.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 2 },
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
// seven judges retain 120s. Supervise all 24.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 },
// Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s.
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS, configuredRetries: 1 },
].map(({ file, attemptMs, caseMs, configuredRetries }) => {
const retries = retriesWithinCaseCap(caseMs, configuredRetries);
return {
file, attemptMs, caseMs, retries,
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
};
}),
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS },
].map(({ file, attemptMs, caseMs }) => ({
file, attemptMs, caseMs,
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: attemptMs + SHARD_RESERVE_MS,
})),
];
/** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */
+201 -1
View File
@@ -14,8 +14,20 @@ import {
formatComparison,
generateCommentary,
judgePassed,
CONTRACT_VIOLATIONS_FILE,
ContractViolation,
TRIAL_ENV,
TRIAL_OUTCOME_SCHEMA,
expectContract,
failureClassOf,
formatTrialOutcomes,
panelVerdict,
parseTrialOutcomes,
sanitizeTrialError,
trialContextFromEnv,
} from './eval-store';
import type { EvalResult, EvalTestEntry, ComparisonResult } from './eval-store';
import type { EvalResult, EvalTestEntry, ComparisonResult, PanelTrial, TrialOutcomeRecord } from './eval-store';
import { EVAL_POLICY } from './periodic-exclude-data';
import { manualReviewFixture } from './manual-judge-review-fixture';
let tmpDir: string;
@@ -957,3 +969,191 @@ describe('generateCommentary', () => {
expect(notes.some(n => n.includes('Stable run'))).toBe(true);
});
});
// --- Trials, panel verdicts and contract vetoes (eval reliability policy) ---
const PANEL = EVAL_POLICY.panel;
const pass = (trial: number, extra: Partial<PanelTrial> = {}): PanelTrial => ({ trial, outcome: 'passed', ...extra });
const fail = (trial: number, extra: Partial<PanelTrial> = {}): PanelTrial => ({ trial, outcome: 'failed', ...extra });
const behavior = (trials: PanelTrial[], quarantined = false) =>
panelVerdict({ case: 'case-x', kind: 'behavior', panel: PANEL, trials, quarantined });
describe('panelVerdict', () => {
test('policy constants are the approved pre-registration', () => {
expect(EVAL_POLICY.panel).toEqual({ n: 3, k: 2 });
expect(EVAL_POLICY.quarantine).toEqual({
entry: { rate: 0.95, minTrials: 10 },
exit: { rate: 0.97, minTrials: 10 },
capFraction: 0.1,
expiryWeeklyRuns: 8,
});
expect(EVAL_POLICY.infraRedispatch).toBe(1);
});
test('rule: one trial, any failure fails the lane', () => {
const ok = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 1, k: 1 }, trials: [pass(1)] });
expect(ok).toMatchObject({ status: 'PASS', split: false, failsLane: false, coverage: true, marks: '✓' });
const bad = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 1, k: 1 }, trials: [fail(1)] });
expect(bad).toMatchObject({ status: 'FAIL', failsLane: true, coverage: false, redClass: 'VERDICT', marks: '✗' });
});
test('behavior 3/3 is a clean PASS', () => {
expect(behavior([pass(1), pass(2), pass(3)])).toMatchObject({ status: 'PASS', split: false, passed: 3, failsLane: false });
});
test('behavior 2/3 is a split PASS that shows its failed trial', () => {
const v = behavior([pass(1), fail(2, { exit_reason: 'timeout' }), pass(3)]);
expect(v).toMatchObject({ status: 'PASS', split: true, passed: 2, failed: 1, failsLane: false, coverage: true, marks: '✓✗✓', reason: 'PASS 2/3' });
expect(v.trials[1].exit_reason).toBe('timeout');
});
test('behavior 1/3 and 0/3 fail the lane', () => {
expect(behavior([pass(1), fail(2), fail(3)])).toMatchObject({ status: 'FAIL', failsLane: true, redClass: 'VERDICT' });
expect(behavior([fail(1), fail(2), fail(3)])).toMatchObject({ status: 'FAIL', failsLane: true });
});
test('a contract trial fails the panel even at 2/3', () => {
const v = behavior([pass(1), pass(2), fail(3, { failure_class: 'contract' })]);
expect(v).toMatchObject({ status: 'FAIL', contract: true, failsLane: true, redClass: 'VERDICT', reason: 'contract violation' });
});
test('a missing trial is INCOMPLETE and fails the lane', () => {
const v = behavior([pass(1), pass(3)]);
expect(v).toMatchObject({ status: 'INCOMPLETE', failsLane: true, coverage: false, redClass: 'INCOMPLETE', marks: '✓·✓' });
expect(v.reason).toContain('missing trial t2');
});
test('duplicate or out-of-range trial records are INCOMPLETE, never deduplicated', () => {
expect(behavior([pass(1), pass(2), pass(2), fail(3)]).status).toBe('INCOMPLETE');
expect(behavior([pass(1), pass(2), pass(3), pass(4)]).reason).toContain('unexpected trial t4');
const v = panelVerdict({ case: 'c', kind: 'behavior', panel: { n: 11, k: 6 }, trials: Array.from({ length: 10 }, (_, i) => pass(i + 2)) });
expect(v.reason).toContain('missing trial t1');
});
test('timeout and infra trials count as failed, never passing', () => {
const v = behavior([pass(1), fail(2, { exit_reason: 'timeout' }), fail(3, { failure_class: 'infra' })]);
expect(v).toMatchObject({ status: 'FAIL', passed: 1, failed: 2, failsLane: true, redClass: 'VERDICT' });
const infra = behavior([pass(1), fail(2, { failure_class: 'infra' }), fail(3, { failure_class: 'infra' })]);
expect(infra).toMatchObject({ status: 'FAIL', redClass: 'INFRA' });
expect(failureClassOf({ exit_reason: 'timeout' })).toBe('timeout');
expect(failureClassOf({})).toBe('assertion');
});
test('quarantined: 1/3 does not fail the lane, 0/3 and contract do, no coverage credit', () => {
expect(behavior([pass(1), fail(2), fail(3)], true)).toMatchObject({ status: 'FAIL', failsLane: false, coverage: false, redClass: null });
expect(behavior([fail(1), fail(2), fail(3)], true)).toMatchObject({ status: 'FAIL', failsLane: true });
expect(behavior([pass(1), pass(2), fail(3, { failure_class: 'contract' })], true)).toMatchObject({ status: 'FAIL', failsLane: true });
expect(behavior([pass(1), pass(2), pass(3)], true)).toMatchObject({ status: 'PASS', coverage: false, failsLane: false });
expect(behavior([pass(1), pass(2)], true)).toMatchObject({ status: 'INCOMPLETE', failsLane: true });
});
test('quarantined rule keeps rule meaning (k = n)', () => {
const v = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 3, k: 3 }, trials: [pass(1), pass(2), fail(3)], quarantined: true });
expect(v).toMatchObject({ status: 'FAIL', failsLane: false });
});
test('all-skipped panel is SKIPPED with no credit; partly skipped is INCOMPLETE', () => {
const skip = (trial: number): PanelTrial => ({ trial, outcome: 'skipped' });
expect(behavior([skip(1), skip(2), skip(3)])).toMatchObject({ status: 'SKIPPED', coverage: false, failsLane: false });
expect(behavior([pass(1), pass(2), skip(3)])).toMatchObject({ status: 'INCOMPLETE', failsLane: true });
});
test('trials of different run attempts are never merged into one verdict', () => {
expect(() => behavior([pass(1), pass(2), fail(3, { attempt: 2 })])).toThrow(/run attempts/);
expect(behavior([pass(1, { attempt: 2 }), pass(2, { attempt: 2 }), pass(3, { attempt: 2 })]).attempt).toBe(2);
});
test('invalid panels throw', () => {
expect(() => panelVerdict({ case: 'c', kind: 'behavior', panel: { n: 3, k: 4 }, trials: [] })).toThrow(/invalid panel/);
expect(() => panelVerdict({ case: 'c', kind: 'nope' as any, panel: { n: 1, k: 1 }, trials: [] })).toThrow(/unknown kind/);
});
});
describe('trial context and expectContract', () => {
let dir: string;
const saved: Record<string, string | undefined> = {};
const keys = [...Object.values(TRIAL_ENV), 'GSTACK_EVAL_DIR'];
beforeEach(() => {
dir = fs.mkdtempSync(path.join(os.tmpdir(), 'panel-verdict-'));
for (const key of keys) saved[key] = process.env[key];
});
afterEach(() => {
for (const key of keys) {
if (saved[key] === undefined) delete process.env[key];
else process.env[key] = saved[key];
}
fs.rmSync(dir, { recursive: true, force: true });
});
const setTrial = () => Object.assign(process.env, {
[TRIAL_ENV.caseId]: 'case-x', [TRIAL_ENV.kind]: 'behavior', [TRIAL_ENV.trial]: '2',
[TRIAL_ENV.panelN]: '3', [TRIAL_ENV.panelK]: '2', [TRIAL_ENV.policyVersion]: String(EVAL_POLICY.version),
GSTACK_EVAL_DIR: dir,
});
test('trialContextFromEnv: absent, complete, and malformed', () => {
for (const key of Object.values(TRIAL_ENV)) delete process.env[key];
expect(trialContextFromEnv()).toBeNull();
setTrial();
expect(trialContextFromEnv()).toEqual({ case_id: 'case-x', kind: 'behavior', trial: 2, panel: { n: 3, k: 2 }, policy_version: EVAL_POLICY.version });
process.env[TRIAL_ENV.trial] = '4';
expect(() => trialContextFromEnv()).toThrow(/Malformed trial context/);
});
test('passing contract is a no-op', () => {
setTrial();
expect(() => expectContract(true, 'fine')).not.toThrow();
expect(fs.existsSync(path.join(dir, CONTRACT_VIOLATIONS_FILE))).toBe(false);
});
test('failed contract stamps the recorded entry and the sidecar before throwing', () => {
setTrial();
const collector = new EvalCollector('e2e', dir);
collector.addTest({ name: 'case-x', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 });
expect(() => expectContract(false, 'handoff missing', { collector, name: 'case-x' })).toThrow(ContractViolation);
const partial = JSON.parse(fs.readFileSync(path.join(dir, '_partial-e2e.json'), 'utf-8'));
expect(partial.tests[0]).toMatchObject({ passed: false, failure_class: 'contract', case_id: 'case-x', trial: 2, kind: 'behavior', panel: { n: 3, k: 2 } });
const sidecar = fs.readFileSync(path.join(dir, CONTRACT_VIOLATIONS_FILE), 'utf-8').trim().split('\n').map((l) => JSON.parse(l));
expect(sidecar).toEqual([expect.objectContaining({ case_id: 'case-x', trial: 2, message: 'handoff missing' })]);
});
test('a contract marked before recording stamps the later record, or becomes its own at finalize', async () => {
setTrial();
const collector = new EvalCollector('e2e', dir);
expect(() => expectContract(0, 'no question asked', { collector, name: 'later' })).toThrow('CONTRACT: no question asked');
collector.addTest({ name: 'later', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 });
expect(() => expectContract(null, 'never recorded', { collector, name: 'orphan' })).toThrow();
const file = await collector.finalize();
const tests = JSON.parse(fs.readFileSync(file, 'utf-8')).tests;
expect(tests.find((t: any) => t.name === 'later')).toMatchObject({ passed: false, failure_class: 'contract' });
expect(tests.find((t: any) => t.name === 'orphan')).toMatchObject({ passed: false, failure_class: 'contract', error: 'never recorded' });
});
});
describe('trial-outcomes JSONL', () => {
const record = (extra: Partial<TrialOutcomeRecord> = {}): TrialOutcomeRecord => ({
schema: TRIAL_OUTCOME_SCHEMA, case: 'case-x', file: 'test/x.test.ts', tier: 'gate', kind: 'behavior',
trial: 1, panel: { n: 3, k: 2 }, attempt: 1, outcome: 'passed', duration_ms: 10, cost_usd: 0.1,
policy_version: EVAL_POLICY.version, quarantined: false, execution: 'executed', source: 'shard', ...extra,
});
test('round-trips valid records', () => {
const records = [record(), record({ trial: 2, outcome: 'failed', failure_class: 'timeout', exit_reason: 'timeout' })];
expect(parseTrialOutcomes(formatTrialOutcomes(records))).toEqual({ records, errors: [] });
});
test('writer fails closed; reader reports bad lines as data errors', () => {
expect(() => formatTrialOutcomes([record({ outcome: 'failed' })])).toThrow(/failed without failure_class/);
expect(() => formatTrialOutcomes([record({ trial: 4 })])).toThrow(/trial invalid/);
const text = `${JSON.stringify(record())}\nnot json\n${JSON.stringify({ ...record(), schema: 'other' })}\n`;
const parsed = parseTrialOutcomes(text);
expect(parsed.records).toHaveLength(1);
expect(parsed.errors).toEqual(['line 2: not JSON', 'line 3: schema other']);
expect(parseTrialOutcomes(text, { maxBytes: 10 }).errors[0]).toContain('exceed');
});
test('sanitizeTrialError keeps one capped line without mentions', () => {
expect(sanitizeTrialError('\n expected @garrytan to `see`\nsecond')).toBe("expected @\u200bgarrytan to 'see'");
expect(sanitizeTrialError('x'.repeat(1000))!.length).toBe(300);
expect(sanitizeTrialError('')).toBeUndefined();
});
});
+367 -1
View File
@@ -76,6 +76,18 @@ export interface EvalTestEntry {
* its body again and re-records under the same name. Set by addTest. */
attempt?: number;
// Trial identity (eval reliability policy). Stamped by addTest from the
// TRIAL_ENV variables the paid runner sets on an isolated trial shard.
/** Registry id (E2E_TIERS / LLM_JUDGE_TOUCHFILES key) this record belongs to. */
case_id?: string;
kind?: EvalCaseKind;
/** 1-based trial index within the case's panel. */
trial?: number;
panel?: PanelShape;
/** Why a failed record failed; 'contract' comes only from expectContract. */
failure_class?: TrialFailureClass;
policy_version?: number;
// E2E
transcript?: any[];
prompt?: string;
@@ -131,6 +143,329 @@ export function evalEntryOutcome(entry: unknown): 'passed' | 'failed' | 'manual-
return result.passed === true ? 'passed' : 'failed';
}
// --- Trials and panel verdicts ---
//
// Paid evals never retry. Each case's kind (E2E_KINDS) fixes its trials before
// the run; a panel verdict is computed once, by panelVerdict(), from exactly
// panel.n trial records of one run attempt. The report, collector-outcomes,
// the PR comment and pass-rates all read that one function.
export type EvalCaseKind = 'rule' | 'behavior' | 'judge';
/** assertion: an ordinary failed expectation. contract: expectContract() fired
* (fails the panel at any count). timeout: the case budget ran out.
* infra: API/CLI/runner failure before the model could be graded. */
export type TrialFailureClass = 'assertion' | 'contract' | 'timeout' | 'infra';
export type TrialOutcome = 'passed' | 'failed' | 'skipped';
export interface PanelShape { n: number; k: number }
/** Environment the paid runner sets on an isolated trial shard. */
export const TRIAL_ENV = {
caseId: 'GSTACK_EVAL_CASE_ID',
kind: 'GSTACK_EVAL_KIND',
trial: 'GSTACK_EVAL_TRIAL',
panelN: 'GSTACK_EVAL_PANEL_N',
panelK: 'GSTACK_EVAL_PANEL_K',
policyVersion: 'GSTACK_EVAL_POLICY_VERSION',
} as const;
/** Sidecar every expectContract() failure appends to (in GSTACK_EVAL_DIR), so a
* contract veto survives a test that throws before recording its entry. */
export const CONTRACT_VIOLATIONS_FILE = 'contract-violations.jsonl';
export interface TrialContext {
case_id: string;
kind: EvalCaseKind;
trial: number;
panel: PanelShape;
policy_version: number;
}
const EVAL_KINDS: readonly EvalCaseKind[] = ['rule', 'behavior', 'judge'];
const FAILURE_CLASSES: readonly TrialFailureClass[] = ['assertion', 'contract', 'timeout', 'infra'];
function positiveInt(raw: string | undefined): number | null {
if (raw === undefined || !/^[1-9][0-9]*$/.test(raw)) return null;
return Number(raw);
}
/** Trial context of this process, or null outside an isolated trial shard.
* A partial or malformed context throws: a mislabeled trial is fail-open. */
export function trialContextFromEnv(env: NodeJS.ProcessEnv = process.env): TrialContext | null {
const caseId = env[TRIAL_ENV.caseId];
if (!caseId) return null;
const kind = env[TRIAL_ENV.kind] as EvalCaseKind | undefined;
const trial = positiveInt(env[TRIAL_ENV.trial]);
const n = positiveInt(env[TRIAL_ENV.panelN]);
const k = positiveInt(env[TRIAL_ENV.panelK]);
const policy = positiveInt(env[TRIAL_ENV.policyVersion]);
if (!kind || !EVAL_KINDS.includes(kind) || trial === null || n === null || k === null || policy === null || k > n || trial > n) {
throw new Error(`Malformed trial context for ${caseId}: ${Object.values(TRIAL_ENV).map((name) => `${name}=${env[name] ?? ''}`).join(' ')}`);
}
return { case_id: caseId, kind, trial, panel: { n, k }, policy_version: policy };
}
/** Failure class of a failed record: an explicit class wins, then the exit reason. */
export function failureClassOf(entry: Pick<EvalTestEntry, 'failure_class' | 'exit_reason'>): TrialFailureClass {
if (entry.failure_class && FAILURE_CLASSES.includes(entry.failure_class)) return entry.failure_class;
return entry.exit_reason === 'timeout' ? 'timeout' : 'assertion';
}
export class ContractViolation extends Error {
constructor(message: string) {
super(`CONTRACT: ${message}`);
this.name = 'ContractViolation';
}
}
/**
* Assert a contract: an outcome the product must meet on every run. On failure
* it records failure_class 'contract' before throwing, both on the collector
* entry named `record.name` (now or when the test records it) and in the
* GSTACK_EVAL_DIR sidecar, so panelVerdict() fails the panel even at 2 of 3.
*/
export function expectContract(
condition: unknown,
message: string,
record?: { collector: EvalCollector | null; name: string },
): asserts condition {
if (condition) return;
record?.collector?.markContractViolation(record.name, message);
const evalDir = process.env.GSTACK_EVAL_DIR;
if (evalDir) {
const context = trialContextFromEnv();
fs.mkdirSync(evalDir, { recursive: true });
fs.appendFileSync(path.join(evalDir, CONTRACT_VIOLATIONS_FILE), JSON.stringify({
case_id: context?.case_id ?? record?.name ?? null,
name: record?.name ?? null,
trial: context?.trial ?? null,
message,
at: new Date().toISOString(),
}) + '\n');
}
throw new ContractViolation(message);
}
export interface PanelTrial {
trial: number;
outcome: TrialOutcome;
/** Required meaning for a failed trial; absent reads as 'assertion'. */
failure_class?: TrialFailureClass;
/** CI run attempt (github.run_attempt); absent means 1. */
attempt?: number;
exit_reason?: string;
error?: string;
execution?: 'executed' | 'reused';
}
export interface PanelVerdictInput {
case: string;
kind: EvalCaseKind;
panel: PanelShape;
trials: readonly PanelTrial[];
quarantined?: boolean;
}
export type PanelStatus = 'PASS' | 'FAIL' | 'INCOMPLETE' | 'SKIPPED';
export interface PanelVerdict {
case: string;
kind: EvalCaseKind;
panel: PanelShape;
attempt: number;
quarantined: boolean;
status: PanelStatus;
passed: number;
failed: number;
/** A failed trial carried failure_class 'contract'. */
contract: boolean;
/** PASS with at least one failed trial: shown as `PASS k/n`, never clean. */
split: boolean;
/** Whether this verdict makes the lane red. */
failsLane: boolean;
/** Whether it counts as passing coverage (never for quarantined or skipped). */
coverage: boolean;
/** Machine classification of a lane-failing verdict: INCOMPLETE (missing or
* malformed trial records), INFRA (every failed trial is infra-class), or
* VERDICT (a real red). Null when the verdict does not fail the lane. */
redClass: 'INCOMPLETE' | 'INFRA' | 'VERDICT' | null;
/** One glyph per trial index: ✓ pass, ✗ fail, – skipped, · missing. */
marks: string;
reason: string;
trials: PanelTrial[];
}
/**
* The single verdict function. `rule`/`judge` cases run panel {1,1}; `behavior`
* cases run EVAL_POLICY.panel; a quarantined case runs a full panel whose k
* keeps its kind's meaning (k = n for rule). Verdict: INCOMPLETE unless
* exactly one record per trial index 1..n; SKIPPED when every trial skipped;
* FAIL on any contract trial; otherwise PASS iff passed >= k. A quarantined
* FAIL fails the lane only on a hard break (0 of n) or a contract violation.
*/
export function panelVerdict(input: PanelVerdictInput): PanelVerdict {
const { n, k } = input.panel;
if (!Number.isInteger(n) || !Number.isInteger(k) || n < 1 || k < 1 || k > n) {
throw new Error(`${input.case}: invalid panel {n:${n}, k:${k}}`);
}
if (!EVAL_KINDS.includes(input.kind)) throw new Error(`${input.case}: unknown kind ${String(input.kind)}`);
const attempts = new Set(input.trials.map((t) => t.attempt ?? 1));
if (attempts.size > 1) {
throw new Error(`${input.case}: trials from run attempts ${[...attempts].join(', ')}; compute one verdict per attempt`);
}
const attempt = [...attempts][0] ?? 1;
const quarantined = input.quarantined === true;
const trials = [...input.trials].sort((a, b) => a.trial - b.trial);
const byIndex = new Map<number, PanelTrial>();
const problems: string[] = [];
for (const t of trials) {
if (!Number.isInteger(t.trial) || t.trial < 1 || t.trial > n) problems.push(`unexpected trial t${t.trial}`);
else if (byIndex.has(t.trial)) problems.push(`duplicate trial t${t.trial}`);
else if (t.outcome !== 'passed' && t.outcome !== 'failed' && t.outcome !== 'skipped') problems.push(`t${t.trial} has outcome ${String(t.outcome)}`);
else byIndex.set(t.trial, t);
}
for (let i = 1; i <= n; i++) if (!trials.some((t) => t.trial === i)) problems.push(`missing trial t${i}`);
const marks = Array.from({ length: n }, (_, i) => {
const t = byIndex.get(i + 1);
return !t ? '·' : t.outcome === 'passed' ? '✓' : t.outcome === 'failed' ? '✗' : '–';
}).join('');
const passed = [...byIndex.values()].filter((t) => t.outcome === 'passed').length;
const failedTrials = [...byIndex.values()].filter((t) => t.outcome === 'failed');
const skipped = [...byIndex.values()].filter((t) => t.outcome === 'skipped').length;
const contract = failedTrials.some((t) => failureClassOf(t) === 'contract');
const base = { case: input.case, kind: input.kind, panel: { n, k }, attempt, quarantined, passed, failed: failedTrials.length, contract, marks, trials };
if (problems.length === 0 && skipped === n) {
return { ...base, status: 'SKIPPED', split: false, failsLane: false, coverage: false, redClass: null, reason: 'every trial skipped (no verdict credit)' };
}
if (problems.length === 0 && skipped > 0) problems.push(`${skipped} of ${n} trials skipped`);
if (problems.length > 0) {
return { ...base, status: 'INCOMPLETE', split: false, failsLane: true, coverage: false, redClass: 'INCOMPLETE', reason: problems.join('; ') };
}
if (!contract && passed >= k) {
const split = failedTrials.length > 0;
return {
...base, status: 'PASS', split, failsLane: false, coverage: !quarantined, redClass: null,
reason: split ? `PASS ${passed}/${n}` : `${passed}/${n} passed`,
};
}
const hardBreak = passed === 0;
const failsLane = !quarantined || contract || hardBreak;
const allInfra = !contract && failedTrials.length > 0 && failedTrials.every((t) => failureClassOf(t) === 'infra');
const why = contract ? 'contract violation' : `${passed}/${n} passed, needs ${k}`;
return {
...base, status: 'FAIL', split: false, failsLane, coverage: false,
redClass: failsLane ? (allInfra ? 'INFRA' : 'VERDICT') : null,
reason: !quarantined ? why
: contract ? `${why}; quarantine never excuses a contract`
: hardBreak ? `${why}; quarantined hard break`
: `${why}; quarantined, does not fail the lane`,
};
}
// --- trial-outcomes JSONL (one line per trial; pass-rate history input) ---
export const TRIAL_OUTCOME_SCHEMA = 'gstack-trial-outcome/v1';
export const TRIAL_OUTCOMES_FILE = 'trial-outcomes.jsonl';
/** Cap on a stored `error` line (sanitized first line of the failure). */
export const TRIAL_ERROR_MAX = 300;
export interface TrialOutcomeRecord {
schema: typeof TRIAL_OUTCOME_SCHEMA;
/** Registry id. */
case: string;
file: string;
tier: string;
kind: EvalCaseKind;
trial: number;
panel: PanelShape;
/** CI run attempt (github.run_attempt); 1 locally and for pre-policy backfill. */
attempt: number;
outcome: TrialOutcome;
/** Present exactly when outcome is 'failed'. */
failure_class?: TrialFailureClass;
exit_reason?: string;
error?: string;
duration_ms: number;
cost_usd: number;
model?: string;
cli_version?: string;
/** Reuse input key of the trial's shard, when known. */
input_identity?: string;
/** EVAL_POLICY.version; 0 marks pre-policy backfill. */
policy_version: number;
quarantined: boolean;
execution: 'executed' | 'reused';
/** shard: isolated trial shard status. junit: a rule file shard's per-test
* JUnit outcome. backfill: imported pre-policy artifact record. */
source: 'shard' | 'junit' | 'backfill';
run_id?: string;
sha?: string;
lane?: string;
recorded_at?: string;
/** History series key: a hash of the case's own touchfiles (GLOBAL_TOUCHFILES excluded), stamped by the report job. */
series_identity?: string;
}
/** First line of free text, stripped of @-mentions and control characters, capped. */
export function sanitizeTrialError(text: string | undefined): string | undefined {
if (!text) return undefined;
const first = text.split('\n').map((l) => l.trim()).find((l) => l.length > 0);
if (!first) return undefined;
// eslint-disable-next-line no-control-regex
const clean = first.replace(/[\u0000-\u001f\u007f]/g, ' ').replace(/`/g, "'").replace(/@(?=[A-Za-z0-9_-])/g, '@\u200b');
return clean.length > TRIAL_ERROR_MAX ? `${clean.slice(0, TRIAL_ERROR_MAX - 1)}…` : clean;
}
function trialRecordProblems(r: any): string[] {
const problems: string[] = [];
if (!r || typeof r !== 'object' || Array.isArray(r)) return ['not an object'];
if (r.schema !== TRIAL_OUTCOME_SCHEMA) problems.push(`schema ${String(r.schema)}`);
for (const key of ['case', 'file', 'tier'] as const) if (typeof r[key] !== 'string' || r[key].length === 0) problems.push(`${key} missing`);
if (!EVAL_KINDS.includes(r.kind)) problems.push(`kind ${String(r.kind)}`);
const n = r.panel?.n, k = r.panel?.k;
if (!Number.isInteger(n) || !Number.isInteger(k) || n < 1 || k < 1 || k > n) problems.push('panel invalid');
if (!Number.isInteger(r.trial) || r.trial < 1 || (Number.isInteger(n) && r.trial > n)) problems.push('trial invalid');
if (!Number.isInteger(r.attempt) || r.attempt < 1) problems.push('attempt invalid');
if (!['passed', 'failed', 'skipped'].includes(r.outcome)) problems.push(`outcome ${String(r.outcome)}`);
if (r.outcome === 'failed' && !FAILURE_CLASSES.includes(r.failure_class)) problems.push('failed without failure_class');
if (r.outcome !== 'failed' && r.failure_class !== undefined) problems.push('failure_class on a non-failed trial');
if (typeof r.duration_ms !== 'number' || !Number.isFinite(r.duration_ms) || r.duration_ms < 0) problems.push('duration_ms invalid');
if (typeof r.cost_usd !== 'number' || !Number.isFinite(r.cost_usd) || r.cost_usd < 0) problems.push('cost_usd invalid');
if (!Number.isInteger(r.policy_version) || r.policy_version < 0) problems.push('policy_version invalid');
if (typeof r.quarantined !== 'boolean') problems.push('quarantined invalid');
if (r.execution !== 'executed' && r.execution !== 'reused') problems.push('execution invalid');
if (!['shard', 'junit', 'backfill'].includes(r.source)) problems.push('source invalid');
if (r.error !== undefined && (typeof r.error !== 'string' || r.error.length > TRIAL_ERROR_MAX)) problems.push('error invalid');
if (r.series_identity !== undefined && (typeof r.series_identity !== 'string' || !/^[\w.-]{1,64}$/.test(r.series_identity))) problems.push('series_identity invalid');
return problems;
}
/** Serialize records as JSONL; throws on any invalid record (writers fail closed). */
export function formatTrialOutcomes(records: readonly TrialOutcomeRecord[]): string {
return records.map((r) => {
const problems = trialRecordProblems(r);
if (problems.length > 0) throw new Error(`invalid trial record ${r?.case}~t${r?.trial}: ${problems.join(', ')}`);
return JSON.stringify(r);
}).join('\n') + (records.length > 0 ? '\n' : '');
}
/** Parse downloaded JSONL as data only: invalid lines are reported, never guessed. */
export function parseTrialOutcomes(text: string, opts: { maxBytes?: number } = {}): { records: TrialOutcomeRecord[]; errors: string[] } {
const maxBytes = opts.maxBytes ?? 16 * 1024 * 1024;
if (Buffer.byteLength(text) > maxBytes) return { records: [], errors: [`trial outcomes exceed ${maxBytes} bytes`] };
const records: TrialOutcomeRecord[] = [];
const errors: string[] = [];
text.split('\n').forEach((line, i) => {
if (line.trim() === '') return;
let parsed: unknown;
try { parsed = JSON.parse(line); } catch { errors.push(`line ${i + 1}: not JSON`); return; }
const problems = trialRecordProblems(parsed);
if (problems.length > 0) errors.push(`line ${i + 1}: ${problems.join(', ')}`);
else records.push(parsed as TrialOutcomeRecord);
});
return { records, errors };
}
export interface EvalResult {
schema_version: number;
version: string;
@@ -887,6 +1222,7 @@ export class EvalCollector {
private shard: string | null;
private fileNamespace?: string;
private createdAt = Date.now();
private pendingContract = new Map<string, string>();
constructor(tier: 'e2e' | 'llm-judge', evalDir?: string, fileNamespace?: string) {
if (fileNamespace !== undefined && !/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(fileNamespace)) {
@@ -903,7 +1239,29 @@ export class EvalCollector {
// names are unique by convention). Stamp the 1-based attempt so a
// pass-on-attempt-2 stays visible forever — the stream hides it.
const prior = this.tests.filter((t) => t.name === entry.name).length;
this.tests.push({ ...entry, attempt: prior + 1 });
const context = trialContextFromEnv();
const record: EvalTestEntry = { ...(context ?? {}), ...entry, attempt: prior + 1 };
const contract = this.pendingContract.get(entry.name);
if (contract !== undefined) {
this.pendingContract.delete(entry.name);
Object.assign(record, { passed: false, failure_class: 'contract', error: record.error ?? contract });
}
this.tests.push(record);
this.savePartial();
}
/** expectContract() hook: mark `name`'s latest record (or its next one) as a
* contract failure. An unmatched mark becomes its own failed record at
* finalize, so the veto is never lost. */
markContractViolation(name: string, message: string): void {
const existing = this.tests.filter((t) => t.name === name).at(-1);
if (!existing) {
this.pendingContract.set(name, message);
return;
}
existing.passed = false;
existing.failure_class = 'contract';
existing.error = existing.error ?? message;
this.savePartial();
}
@@ -959,6 +1317,14 @@ export class EvalCollector {
async finalize(): Promise<string> {
if (this.finalized) return '';
this.finalized = true;
for (const [name, message] of this.pendingContract) {
this.tests.push({
...(trialContextFromEnv() ?? {}),
name, suite: 'contract', tier: this.tier, passed: false, duration_ms: 0, cost_usd: 0,
failure_class: 'contract', error: message, attempt: 1,
});
}
this.pendingContract.clear();
const git = getGitInfo();
const version = getVersion();
+59
View File
@@ -23,6 +23,8 @@ export interface JudgeScore {
reasoning: string;
}
export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const;
export interface JudgeRefusalEvidence {
stop_reason: 'refusal';
response_id: string | null;
@@ -196,6 +198,63 @@ export async function callJudge<T>(
}
}
/**
* Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this
* helper (imported by many paid tests) does not pull the quarantine registry
* into their touchfile closure. test/judge-panel.test.ts pins the two equal.
*/
export const JUDGE_PANEL_SAMPLES = 3;
/**
* Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of
* independent samples of the SAME prompt concurrently, inside its unchanged
* JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
* against the unchanged minimum; boolean fields gate on a strict majority.
* A sample that errors (refusal, truncation, non-JSON, malformed field) fails
* the whole panel and is never resampled. callJudge's 429 backoff happens
* before any model output exists, so it is transport, not a verdict retry.
*/
export async function judgePanel<T>(sample: () => Promise<T>): Promise<T[]> {
const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample()));
const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []);
if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult<T>).value);
const first = failures[0]!;
// A refusal is an unscored panel only when EVERY sample refused; a partial
// refusal beside scored samples is an ordinary failed panel.
if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) {
throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`);
}
throw first.reason;
}
/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */
export function judgePanelMean<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, keys: readonly K[]): Record<K, number> {
if (samples.length === 0) throw new Error('Judge panel has no samples');
return Object.fromEntries(keys.map(key => {
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value));
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`);
return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length];
})) as Record<K, number>;
}
/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */
export function judgePanelMajority<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, key: K): boolean {
if (samples.length === 0) throw new Error('Judge panel has no samples');
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
const bad = values.findIndex(value => typeof value !== 'boolean');
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`);
return values.filter(value => value === true).length * 2 > values.length;
}
/** Sample reasoning lines, numbered, for the collector record. */
export function judgePanelReasoning(samples: ReadonlyArray<unknown>): string {
return samples.map((sample, index) => {
const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined;
return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`;
}).join('\n');
}
/**
* Score documentation quality on clarity/completeness/actionability (1-5).
*/
+64
View File
@@ -59,3 +59,67 @@ export const CASE_CI_EXCLUDE: Record<string, { reason: string; tracking: string
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
};
/**
* Paid-eval verdict policy, pre-registered (approved 2026-09-29). Frozen before
* the census: any change after seeing census results needs Garry's
* re-approval and a fresh census, and bumps `version` (every trial record
* carries it as policy_version, so pass-rate history segments at the change).
* panel - behavior cases and quarantined cases run n independent
* trials; a behavior panel PASSES at >= k passing trials with
* no contract violation. Rule and judge cases run one trial.
* quarantine - entry below `entry.rate` per trial over >= `entry.minTrials`
* new-policy trials; exit at >= `exit.rate` over >=
* `exit.minTrials`; at most `capFraction` of each tier's
* blocking cases; an entry expires after `expiryWeeklyRuns`.
* judge - a judge case draws `samples` independent samples of one
* prompt concurrently; numeric dimensions gate on the panel
* mean against the unchanged threshold, booleans on a strict
* majority; an erroring sample fails the panel, never resampled.
* drift - one-sided Fisher exact alarm between input-identity series
* (Holm-controlled across the cases tested in one report).
* infraRedispatch - a census whose every red verdict is machine-classified
* INFRA or INCOMPLETE may be re-dispatched this many times as
* a new run; both runs are reported.
*/
export const EVAL_POLICY = {
version: 1,
panel: { n: 3, k: 2 },
quarantine: {
entry: { rate: 0.95, minTrials: 10 },
exit: { rate: 0.97, minTrials: 10 },
capFraction: 0.10,
expiryWeeklyRuns: 8,
},
judge: { samples: 3 },
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
infraRedispatch: 1,
} as const;
/**
* Quarantined paid cases, keyed by registry id (an E2E_TIERS key). A
* quarantined case still runs its full panel and reports in every lane, but
* its failed verdict cannot fail the lane unless the panel is a hard break
* (0 of n) or a trial violated a contract; it never counts as passing
* coverage. An entry needs the entry rule met on the current input identity,
* a written diagnosis that the failures are detector, harness or model-latency
* failures (a product defect is never quarantined), and unchanged case
* touchfiles in the change that adds it. Pinned by
* test/periodic-exclude-policy.test.ts.
* reason - the written diagnosis, with the pass-rate evidence
* failureClass - what the diagnosis found; a product defect has no class here
* tracking - issue or TODOS pointer
* owner - who removes it
* enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here)
* exit - the measurable exit condition
* At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and
* periodic are the blocking tiers) may be quarantined at once.
*/
export const CASE_QUARANTINE: Record<string, {
reason: string;
failureClass: 'detector' | 'harness' | 'model-latency';
tracking: string;
owner: string;
enteredAt: string;
exit: string;
}> = {};
+16 -3
View File
@@ -34,6 +34,8 @@ import {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
} from './touchfiles-data';
/** Repo-relative path of the pure-data file (the map-diff subject). */
@@ -145,6 +147,9 @@ export interface TouchfileMaps {
E2E_TIERS: Record<string, string>;
LLM_JUDGE_TOUCHFILES: Record<string, string[]>;
GLOBAL_TOUCHFILES: string[];
/** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */
E2E_KINDS?: Record<string, string>;
BEHAVIOR_WHY?: Record<string, string>;
}
export type MapDiffCause =
@@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
};
function isStringArray(v: unknown): v is string[] {
@@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps {
return isRecordOfStringArrays(o.E2E_TOUCHFILES)
&& isRecordOfStrings(o.E2E_TIERS)
&& isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES)
&& isStringArray(o.GLOBAL_TOUCHFILES);
&& isStringArray(o.GLOBAL_TOUCHFILES)
&& (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS))
&& (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY));
}
/**
* Pure map-diff core (injectable for tests — no git, no filesystem).
*
* A key counts as CHANGED when it was added to any per-key map, its dep-list
* array differs, or its tier value flipped. A key counts as REMOVED only when
* array differs, or its tier, kind or behavior tolerance changed. A per-key
* map missing on the old side (a base revision older than E2E_KINDS /
* BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when
* it is gone from every new per-key map; a key dropped from one map but still
* present in another (e.g. tier entry deleted, touchfile entry kept) counts
* as changed — conservative, because the test still exists with a different
@@ -212,7 +223,7 @@ export function diffTouchfileMapsCore(
oldMaps: TouchfileMaps,
newMaps: TouchfileMaps,
): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } {
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const;
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const;
const changed = new Set<string>();
const rawRemoved = new Set<string>();
@@ -290,6 +301,8 @@ export function diffTouchfileMaps(
' E2E_TIERS: m.E2E_TIERS,',
' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,',
' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,',
' E2E_KINDS: m.E2E_KINDS,',
' BEHAVIOR_WHY: m.BEHAVIOR_WHY,',
'}));',
'',
].join('\n'));
+303
View File
@@ -1573,3 +1573,306 @@ export const GLOBAL_TOUCHFILES = [
// diffed per key, so a data-only edit runs just the affected tests.
// Map-diff fails CLOSED — any error on that path still runs everything.
];
/**
* Eval kind per live case (every E2E_TIERS and LLM_JUDGE_TOUCHFILES key).
* The kind fixes the trial policy before the run (EVAL_POLICY in
* periodic-exclude-data.ts):
* rule - one trial; any failed assertion fails the verdict. The default.
* behavior - a panel of independent trials, PASS at the policy majority;
* needs a BEHAVIOR_WHY entry naming the tolerated deviation.
* judge - an LLM-judge score of a static input, sampled as a panel.
* Reclassification is a reviewed diff, never a runtime switch.
*/
export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'ship-skipped-queued-finding': 'rule',
'investigate-owned-completion': 'rule',
'investigate-owned-abort': 'rule',
'investigate-owned-ending-error': 'rule',
'shared-libs-review-path-eligibility': 'rule',
'shared-libs-review-index-flags': 'rule',
'shared-libs-review-prior-coverage': 'rule',
'shared-libs-codex-read-only': 'rule',
'shared-libs-read-only': 'rule',
'shared-libs-unsupported-git': 'rule',
'shared-libs-review-lifecycle': 'rule',
'shared-libs-review-revalidation': 'rule',
'shared-libs-opportunity-judgment': 'behavior',
'shared-libs-pr-coverage': 'rule',
'shared-libs-plan-callers': 'rule',
'browse-basic': 'rule',
'browse-snapshot': 'rule',
'aside-browse-basic': 'rule',
'aside-browse-flow': 'rule',
'aside-qa-quick': 'rule',
'aside-scrape-json': 'rule',
'aside-canary-quick': 'rule',
'hermetic-canary': 'rule',
'hermetic-sentinel': 'rule',
'skillmd-setup-discovery': 'rule',
'skillmd-no-local-binary': 'rule',
'skillmd-outside-git': 'rule',
'session-awareness': 'rule',
'operational-learning': 'rule',
'first-task-scaffold': 'rule',
'qa-quick': 'rule',
'qa-b6-static': 'rule',
'qa-b7-spa': 'rule',
'qa-b8-checkout': 'rule',
'qa-only-no-fix': 'rule',
'qa-fix-loop': 'rule',
'qa-bootstrap': 'rule',
'review-exploratory-small-cli': 'rule',
'ship-exploratory-small-cli': 'rule',
'ship-exploratory-unavailable': 'rule',
'ship-exploratory-plan-checks': 'rule',
'ship-exploratory-late-input': 'rule',
'qa-functional-cli-report': 'rule',
'qa-functional-webhook-report': 'rule',
'qa-functional-cli-fix': 'rule',
'qa-functional-webhook-fix': 'rule',
'review-sql-injection': 'rule',
'review-enum-completeness': 'rule',
'review-base-branch': 'rule',
'review-design-lite': 'behavior',
'review-coverage-audit': 'rule',
'review-dashboard-via': 'rule',
'review-army-migration-safety': 'rule',
'review-army-perf-n-plus-one': 'rule',
'review-army-delivery-audit': 'rule',
'review-army-quality-score': 'rule',
'review-army-json-findings': 'rule',
'review-army-red-team': 'behavior',
'review-army-consensus': 'behavior',
'review-army-simplification': 'behavior',
'review-army-simplification-precision': 'behavior',
'office-hours-spec-review': 'rule',
'office-hours-brain-writeback': 'behavior',
'gbrain-roundtrip-local': 'rule',
'sync-gbrain-read-ready': 'rule',
'sync-gbrain-read-unknown': 'rule',
'office-hours-forcing-energy': 'behavior',
'office-hours-builder-wildness': 'behavior',
'plan-ceo-review': 'rule',
'plan-ceo-review-selective': 'rule',
'plan-ceo-review-benefits': 'rule',
'plan-ceo-review-expansion-energy': 'behavior',
'plan-eng-review': 'rule',
'plan-eng-review-artifact': 'rule',
'plan-eng-coverage-audit': 'rule',
'plan-review-report': 'rule',
'plan-ceo-review-plan-mode': 'rule',
'plan-eng-review-plan-mode': 'rule',
'plan-design-review-plan-mode': 'rule',
'plan-devex-review-plan-mode': 'rule',
'plan-mode-no-op': 'rule',
'office-hours-auto-mode': 'rule',
'auto-decide-preserved': 'rule',
'auq-format-gate': 'rule',
'plan-ceo-mode-routing': 'rule',
'plan-design-with-ui-scope': 'rule',
'tpa-present': 'rule',
'tpa-absent-linux': 'rule',
'tpa-broken': 'rule',
'tpa-absent-darwin': 'rule',
'tpa-apple-ban': 'rule',
'ship-section-loading': 'rule',
'plan-ceo-section-loading': 'rule',
'carve-section-loading': 'rule',
'plan-eng-finding-floor': 'rule',
'plan-ceo-finding-floor': 'rule',
'plan-design-finding-floor': 'rule',
'plan-devex-finding-floor': 'rule',
'plan-eng-multi-finding-batching': 'rule',
'plan-ceo-split-overflow': 'rule',
'setup-gbrain-remote': 'rule',
'setup-gbrain-bad-token': 'rule',
'setup-gbrain-path4-local-pglite': 'rule',
'plan-ceo-review-format-mode': 'behavior',
'plan-ceo-review-format-approach': 'behavior',
'plan-eng-review-format-coverage': 'behavior',
'plan-eng-review-format-kind': 'behavior',
'office-hours-phase4-fork': 'behavior',
'llm-judge-recommendation': 'judge',
'plan-ceo-review-prosons-cadence': 'behavior',
'plan-review-prosons-format': 'behavior',
'plan-review-prosons-hardstop-neg': 'behavior',
'plan-review-prosons-neutral-neg': 'behavior',
'plan-tune-inspect': 'rule',
'codex-offered-office-hours': 'rule',
'codex-offered-ceo-review': 'rule',
'codex-offered-design-review': 'rule',
'codex-offered-eng-review': 'rule',
'timeline-event-flow': 'rule',
'context-recovery-artifacts': 'rule',
'context-save-writes-file': 'rule',
'context-restore-loads-latest': 'rule',
'context-save-routing': 'rule',
'context-save-then-restore-roundtrip': 'rule',
'context-restore-fragment-match': 'rule',
'context-restore-empty-state': 'rule',
'context-restore-list-delegates': 'rule',
'context-restore-legacy-compat': 'rule',
'context-save-list-current-branch': 'rule',
'context-save-list-all-branches': 'rule',
'ship-base-branch': 'rule',
'ship-local-workflow': 'rule',
'ship-managed-hook-refresh': 'rule',
'ship-unmanaged-hook-consent': 'rule',
'ship-local-hook-preservation': 'rule',
'ship-coverage-audit': 'rule',
'ship-triage': 'rule',
'ship-docsync-missing-marker': 'rule',
'ship-docsync-missing-asset': 'rule',
'ship-docsync-launch-failure': 'rule',
'ship-docsync-timeout-unsettled': 'rule',
'ship-docsync-late-result': 'rule',
'ship-docsync-stale-before': 'rule',
'ship-docsync-stale-after': 'rule',
'ship-docsync-recovery': 'rule',
'ship-docsync-completion': 'rule',
'ship-docsync-current': 'rule',
'ship-docsync-failure': 'rule',
'ship-docsync-store': 'rule',
'docsync-spawned': 'rule',
'retro': 'rule',
'retro-base-branch': 'rule',
'cso-full-audit': 'rule',
'cso-diff-mode': 'rule',
'cso-infra-scope': 'rule',
'learnings-show': 'rule',
'document-release': 'rule',
'codex-review': 'rule',
'codex-discover-skill': 'rule',
'codex-review-findings': 'rule',
'outside-voice-codex-to-claude-code': 'rule',
'outside-voice-claude-code-to-codex': 'rule',
'outside-plan-disabled-no-fallback': 'rule',
'codex-sol-scope-termination': 'rule',
'design-consultation-core': 'rule',
'design-consultation-existing': 'rule',
'design-consultation-research': 'rule',
'design-consultation-preview': 'rule',
'plan-design-review-no-ui-scope': 'rule',
'design-review-fix': 'rule',
'design-review-detector-shim': 'rule',
'design-review-detector-shim-dom': 'rule',
'design-review-plugin-handoff': 'rule',
'design-html-slop-gate': 'behavior',
'diagram-triplet': 'rule',
'diagram-authoring-quality': 'rule',
'gstack-upgrade-happy-path': 'rule',
'land-and-deploy-workflow': 'rule',
'land-and-deploy-first-run': 'rule',
'land-and-deploy-review-gate': 'rule',
'canary-workflow': 'rule',
'benchmark-workflow': 'rule',
'setup-deploy-workflow': 'rule',
'autoplan-dual-voice': 'rule',
'benchmark-providers-live': 'rule',
'scrape-match-path': 'behavior',
'scrape-prototype-path': 'behavior',
'skillify-happy-path': 'rule',
'skillify-provenance-refusal': 'rule',
'skillify-approval-reject': 'rule',
'journey-ideation': 'rule',
'journey-plan-eng': 'rule',
'journey-debug': 'rule',
'journey-qa': 'rule',
'journey-code-review': 'rule',
'journey-ship': 'rule',
'journey-docs': 'rule',
'journey-retro': 'rule',
'journey-design-system': 'rule',
'journey-visual-qa': 'rule',
'ios-qa-device': 'rule',
'arm-benchmark-native-overbuild': 'rule',
'arm-benchmark-crud-endpoint': 'rule',
'arm-benchmark-bugfix-decoys': 'rule',
'office-hours-section-loading': 'rule',
'office-hours-design-draft': 'rule',
'plan-decision-classification': 'rule',
'plan-devex-peer-comparison-classification': 'rule',
'health-reporting': 'rule',
'overlay-harness-claude-dedicated-tools-vs-bash': 'rule',
'overlay-harness-opus-4-7-effort-match-trivial': 'rule',
'overlay-harness-opus-4-7-literal-interpretation': 'rule',
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule',
'journey-negatives': 'rule',
'review/SKILL.md workflow': 'judge',
'setup-browser-cookies/SKILL.md workflow': 'judge',
'browse/SKILL.md reference': 'judge',
'setup block': 'judge',
'qa/SKILL.md workflow': 'judge',
'qa/SKILL.md health rubric': 'judge',
'qa/SKILL.md anti-refusal': 'judge',
'cross-skill greptile consistency': 'judge',
'ship/SKILL.md workflow': 'judge',
'document-release/SKILL.md workflow': 'judge',
'plan-ceo-review/SKILL.md modes': 'judge',
'plan-eng-review/SKILL.md sections': 'judge',
'plan-design-review/SKILL.md passes': 'judge',
'design-review/SKILL.md fix loop': 'judge',
'design-consultation/SKILL.md research': 'judge',
'land-and-deploy/SKILL.md workflow': 'judge',
'canary/SKILL.md monitoring loop': 'judge',
'benchmark/SKILL.md perf collection': 'judge',
'setup-deploy/SKILL.md platform setup': 'judge',
'retro/SKILL.md instructions': 'judge',
'qa-only/SKILL.md workflow': 'judge',
'gstack-upgrade/SKILL.md upgrade flow': 'judge',
'sync-gbrain/SKILL.md read-only readiness': 'judge',
'voice directive tone': 'judge',
};
/**
* One-line tolerance for every behavior-kind case: why an occasional
* deviation is acceptable product behavior. Keys equal the behavior ids of
* E2E_KINDS; values are non-empty.
*/
export const BEHAVIOR_WHY: Record<string, string> = {
'shared-libs-opportunity-judgment':
"Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.",
'review-design-lite':
"How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.",
'review-army-red-team':
"Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.",
'review-army-consensus':
"Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.",
'review-army-simplification':
"Flagging the planted unnecessary structure is an advisory-lens quality judgment.",
'review-army-simplification-precision':
"Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.",
'office-hours-forcing-energy':
"The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.",
'office-hours-builder-wildness':
"Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.",
'office-hours-brain-writeback':
"The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.",
'office-hours-phase4-fork':
"Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.",
'plan-ceo-review-expansion-energy':
"Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.",
'plan-ceo-review-format-mode':
"Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.",
'plan-ceo-review-format-approach':
"Approach-menu Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-coverage':
"Coverage-issue Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-kind':
"Kind-note wording is live formatting of one AskUserQuestion.",
'plan-ceo-review-prosons-cadence':
"Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.",
'plan-review-prosons-format':
"The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.",
'plan-review-prosons-hardstop-neg':
"Not using the hard-stop escape on an ordinary decision is live formatting of one question.",
'plan-review-prosons-neutral-neg':
"Avoiding neutral posture and naming a because-reason is live formatting of one question.",
'design-html-slop-gate':
"How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.",
'scrape-match-path':
"The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.",
'scrape-prototype-path':
"The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.",
};
+24 -11
View File
@@ -4,7 +4,7 @@ import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
import { JUDGE_MS } from './eval-budgets';
import type { JudgeScore } from './llm-judge';
import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input';
import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache,
type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache';
@@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres
|| typeof value.reasoning !== 'string'
|| (structuredResponse && (!value.reasoning.trim()
|| value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false;
return (['clarity', 'completeness', 'actionability'] as const).every(key =>
return JUDGE_SCORE_DIMENSIONS.every(key =>
typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5);
}
const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 };
/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */
export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array<JudgeScore & EvalCacheValue> } {
if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples'
|| !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES
|| !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false;
const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS);
return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]);
}
export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null;
lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null;
/** The attempt guard is rechecked after synchronous input/provenance reads. */
publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined;
publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined;
} {
const env = opts.env ?? process.env;
const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined };
const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined };
const pr = Number(env.EVALS_CACHE_PR);
// Runtime ID is the immutable CI image manifest, not a mutable image tag.
// Nonstandard Node/Bun preload code or custom model endpoints need a separate
@@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
prompts: { [opts.testName]: prompt },
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1,
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0,
panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' },
...(opts.stream ? { stream: true } : {}),
...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) },
@@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
return {
lookup() {
const result = lookupEvalInputCache({ ...common, identity: before,
validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) });
validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) });
return result.status === 'reused'
? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null;
? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null;
},
publish(scores, isActive = () => true) {
publish(samples, isActive = () => true) {
// Caller reaches here ONLY after its actual assertions passed. A later
// failed case in the file does not erase this independently completed case.
if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) };
if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
const after = currentIdentity();
const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID;
if (!after || !runId || !isActive()) return;
@@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
cancelled: false, skipped: 0, failed: 0, passed: 1,
cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }],
source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() },
result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning },
result: panel,
} });
// A slow synchronous write can consume the recording allowance. The
// caller withdraws this new receipt if its final deadline check fails.
+1 -1
View File
@@ -27,7 +27,7 @@ function runnerDependencies(root: string, entries: string[]): string[] {
const source = fs.readFileSync(file, 'utf8').replace(/^#![^\n]*(?:\n|$)/, '\n');
let audited = source;
if (relative === 'test/helpers/test-selection.ts') {
if (createHash('sha256').update(source).digest('hex') !== '4d2fbcb6249e8d22453d25bfe9b18ee0f4568bbec071918675c38a455d4e1e08') {
if (createHash('sha256').update(source).digest('hex') !== '052ad5a52472bcb41db04c9f21fe6a819e9547768468f7e5d390e0014b567677') {
throw new Error('Re-audit the historical touchfile map loader before excluding its computed import');
}
audited = source.replace('`const m = await import(${JSON.stringify(dataPath)});`,', "'',");
+6 -6
View File
@@ -21,8 +21,8 @@ const fakeEnv = {
};
describe('overlay file policy', () => {
test('grouped planning isolates every overlay and preserves ordinary retries', () => {
// Two short-case files keep their one retry (timeout-is-a-verdict rule).
test('grouped planning isolates every overlay and never retries ordinary files', () => {
// Paid evals never retry (approved 2026-09-29), short-case files included.
const workflow = 'test/skill-e2e-review.test.ts';
const files = [...overlayFiles, 'test/skill-e2e-triage.test.ts', workflow];
for (const maxFilesPerShard of [2, 3, 10]) {
@@ -31,9 +31,9 @@ describe('overlay file policy', () => {
for (const file of overlayFiles) expect(shards).toContainEqual([file]);
const workflowShard = shards.find(shard => shard.includes(workflow))!;
expect(workflowShard.some(isOverlayTestFile)).toBe(false);
expect(retriesForFiles(workflowShard)).toBe(1);
expect(retriesForFiles(workflowShard)).toBe(0);
const args = buildPaidShardArgs(workflowShard, resolvePaidShardTimeoutMs(workflowShard), 2, retriesForFiles(workflowShard));
expect(args[args.indexOf('--retry') + 1]).toBe('1');
expect(args[args.indexOf('--retry') + 1]).toBe('0');
expect(planPaidShards(files.map(file => file.replaceAll('/', '\\')), { maxFilesPerShard })).toEqual(shards);
}
});
@@ -66,10 +66,10 @@ describe('overlay file policy', () => {
for (const file of [normalFile, 'test/skill-e2e-overlay-harness.test.ts', 'test/model-overlays.test.ts']) {
expect(isOverlayTestFile(file)).toBe(false);
expect(resolvePaidShardTimeoutMs([file])).toBe(DEFAULT_SHARD_TIMEOUT_MS);
// Not overlays; unlisted files run once because their case budget is unknown.
// Not overlays; every paid file runs once.
expect(retriesForFiles([file])).toBe(0);
}
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1);
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(0);
expect(resolvePaidShardTimeoutMs([normalFile], 1234)).toBe(1234);
expect(resolvePaidShardTimeoutMs([overlayFiles[0]], 1_900_000)).toBe(1_900_000);
expect(() => resolvePaidShardTimeoutMs([overlayFiles[0]], 1_800_000)).toThrow('explicit wall');
+3 -2
View File
@@ -241,7 +241,7 @@ describe('PR profile paid-runner integration', () => {
expect(guarded[0].status).toBe('passed-empty');
});
test('report distinguishes retained/deferred coverage and final executed/reused outcomes from attempts', () => {
test('report distinguishes retained/deferred coverage and counts every executed/reused record', () => {
const manifest = ceoManifest();
const lines = formatProfileCoverage(manifest).join('\n');
expect(lines).toContain('profile=pr mode=pr');
@@ -252,6 +252,7 @@ describe('PR profile paid-runner integration', () => {
{ name: 'retry', suite: 'judge', passed: true, execution: 'executed' },
{ name: 'cached', suite: 'judge', passed: true, execution: 'reused' },
{ name: 'failed', suite: 'native', passed: false },
] }])).toEqual({ executed: 2, reused: 1, passed: 2, failed: 1, manual_accepted: 0, attempts: 4 });
// Paid evals never retry: a later pass never replaces an earlier failed record.
] }])).toEqual({ executed: 3, reused: 1, passed: 2, failed: 2, manual_accepted: 0, attempts: 4 });
});
});
+237
View File
@@ -0,0 +1,237 @@
/**
* Fail-open regression suite for the paid lane verdict. Synthetic slice
* artifacts go through the real `--report` CLI path (the command the workflow
* report jobs run), so a change to the gate cannot turn a real failure green
* without one of these cases going red. Landed before the panel-verdict gate
* change; every later gate change extends it.
*/
import { afterAll, beforeAll, describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { parseRunManifest, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards';
import { stampTrialSeries } from '../scripts/eval-trial-series';
const ROOT = path.resolve(import.meta.dir, '..');
const RULE_A = 'test/skill-e2e-fail-open-alpha.test.ts';
const RULE_B = 'test/skill-e2e-fail-open-beta.test.ts';
let base: string;
beforeAll(() => { base = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-fail-open-')); });
afterAll(() => { fs.rmSync(base, { recursive: true, force: true }); });
type Outcome = SliceResult['outcomes'][number];
const passed = (file: string, extra: Partial<Outcome> = {}): Outcome =>
({ files: [file], status: 'passed', exitCode: 0, elapsedMs: 1_000, executedTests: 1, skippedTests: 0, ...extra });
function manifest(entries: PaidRunManifest['entries'], sliceCount: number): PaidRunManifest {
return parseRunManifest(JSON.stringify({ version: 1, tier: 'periodic', evalsAll: true, sliceCount,
selectionReason: 'fail-open fixture', profile: 'full', selection: { e2e: null, judges: null }, entries }));
}
let caseCounter = 0;
function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record<string, unknown> = {}, env: NodeJS.ProcessEnv = {}) {
const dir = path.join(base, `case-${++caseCounter}`);
fs.mkdirSync(dir, { recursive: true });
fs.writeFileSync(path.join(dir, 'manifest.json'), JSON.stringify(plan));
for (const slice of slices) fs.writeFileSync(path.join(dir, `slice-${slice.sliceIndex}.json`), JSON.stringify(slice));
for (const [name, body] of Object.entries(collectors)) {
fs.mkdirSync(path.dirname(path.join(dir, name)), { recursive: true });
fs.writeFileSync(path.join(dir, name), JSON.stringify(body));
}
const result = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--tier', plan.tier, '--report', dir],
{ cwd: ROOT, encoding: 'utf8', timeout: 30_000, env: { ...process.env, GITHUB_RUN_ID: '', GITHUB_SHA: '', EVALS_TIER: plan.tier, ...env } });
return { status: result.status, out: `${result.stdout}\n${result.stderr}`, dir };
}
const slice = (sliceIndex: number, sliceCount: number, outcomes: Outcome[]): SliceResult =>
({ version: 1, tier: 'periodic', profile: 'full', selection: { e2e: null, judges: null }, sliceIndex, sliceCount, outcomes });
describe('rule shards stay fail-closed through --report', () => {
const plan = manifest([
{ file: RULE_A, slice: 1, status: 'planned' },
{ file: RULE_B, slice: 2, status: 'planned' },
], 2);
test('all planned rule shards passed: green', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A)]), slice(2, 2, [passed(RULE_B)])]);
expect(r.status, r.out).toBe(0);
});
test('a failed rule shard: red, naming the file', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'failed', exitCode: 1 })]), slice(2, 2, [passed(RULE_B)])]);
expect(r.status).toBe(1);
expect(r.out).toContain(`${RULE_A}: failed`);
});
test('a timed-out rule shard: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'timed-out', exitCode: null })]), slice(2, 2, [passed(RULE_B)])]);
expect(r.status).toBe(1);
expect(r.out).toContain(`${RULE_A}: timed-out`);
});
test('a missing slice artifact: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A)])]);
expect(r.status).toBe(1);
expect(r.out).toContain('slice 2/2 reported NO result');
});
test('a planned shard no slice reported: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A)]), slice(2, 2, [])]);
expect(r.status).toBe(1);
expect(r.out).toContain(`planned ${RULE_B} (slice 2) was never reported`);
});
test('a hollow shard under EVALS_ALL: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'passed-empty', executedTests: 0 })]), slice(2, 2, [passed(RULE_B)])]);
expect(r.status).toBe(1);
expect(r.out).toContain(`${RULE_A}: passed-empty`);
});
test('a never-started shard: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'never-started', exitCode: null, executedTests: null })]), slice(2, 2, [passed(RULE_B)])]);
expect(r.status).toBe(1);
});
test('a failed collector record under a passing shard: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A)]), slice(2, 2, [passed(RULE_B)])], {
'shards/skill-e2e-fail-open-alpha/run.json': { tier: 'e2e', total_tests: 1, total_cost_usd: 0,
tests: [{ name: 'alpha', suite: 's', tier: 'e2e', passed: false, duration_ms: 1, cost_usd: 0 }] },
});
expect(r.status).toBe(1);
expect(r.out).toContain('1 unapproved final collector failure(s)');
});
test('a shard reported by the wrong slice: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A), passed(RULE_B)]), slice(2, 2, [])]);
expect(r.status).toBe(1);
expect(r.out).toContain(`${RULE_B} planned for slice 2 but reported by slice 1`);
});
});
describe('behavior and quarantined panels through --report', () => {
const FILE = 'test/skill-e2e-review.test.ts';
const ID = 'review-design-lite';
const key = (trial: number) => `${FILE}#${ID}~t${trial}`;
const plan = (quarantined = false) => manifest([
...[1, 2, 3].map(trial => ({ file: key(trial), slice: trial, status: 'planned' as const,
trial: { kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined } })),
{ file: RULE_A, slice: 4, status: 'planned' },
], 4);
type TrialResult = 'passed' | 'failed' | 'contract' | 'missing' | 'harness';
const trialOutcome = (trial: number, result: TrialResult, quarantined = false): Outcome | null => {
if (result === 'missing') return null;
const record = { case: ID, trial, kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined, cost_usd: 0, duration_ms: 1_000 };
if (result === 'harness') return passed(key(trial), { status: 'never-started', exitCode: null, executedTests: null, skippedTests: null,
trial: { ...record, outcome: null, harness: 'never started' } });
if (result === 'passed') return passed(key(trial), { trial: { ...record, outcome: 'passed' } });
return passed(key(trial), { status: 'failed', exitCode: 1,
trial: { ...record, outcome: 'failed', failure_class: result === 'contract' ? 'contract' : 'timeout',
exit_reason: 'timeout', timeout_at_turn: 14, error: result === 'contract' ? 'handoff missing' : 'no posture match' } });
};
const run = (results: TrialResult[], quarantined = false, dropSlice?: number) => report(plan(quarantined), [1, 2, 3, 4]
.filter(index => index !== dropSlice)
.map(index => slice(index, 4, index === 4 ? [passed(RULE_A)]
: [trialOutcome(index, results[index - 1]!, quarantined)].filter((o): o is Outcome => o !== null))));
test('behavior 3/3: green', () => {
const r = run(['passed', 'passed', 'passed']);
expect(r.status, r.out).toBe(0);
expect(r.out).toContain('VERDICT GREEN');
});
test('behavior 2/3: green, the failed trial shown with its cause', () => {
const r = run(['passed', 'failed', 'passed']);
expect(r.status, r.out).toBe(0);
expect(r.out).toContain(`⚠ ${ID} behavior PASS 2/3 (✓✗✓)`);
expect(r.out).toContain('t2: timeout at turn 14');
const summary = JSON.parse(fs.readFileSync(path.join(r.dir, 'collector-outcomes.json'), 'utf8'));
expect(summary.version).toBe(2);
expect(summary.panels[0]).toMatchObject({ case: ID, status: 'PASS', split: true, failsLane: false });
const outcomesFile = path.join(r.dir, 'trial-outcomes.jsonl');
const history = () => fs.readFileSync(outcomesFile, 'utf8').trim().split('\n').map(line => JSON.parse(line));
expect(history().map(h => [h.trial, h.outcome])).toEqual([[1, 'passed'], [2, 'failed'], [3, 'passed']]);
expect(stampTrialSeries(outcomesFile)).toBe(3);
expect(new Set(history().map(h => h.series_identity)).size).toBe(1);
expect(history()[0].series_identity).toMatch(/^[0-9a-f]{16}$/);
});
test('behavior 1/3: red', () => {
const r = run(['passed', 'failed', 'failed']);
expect(r.status).toBe(1);
expect(r.out).toContain(`PANEL ${ID} FAIL 1/3`);
});
test('a missing trial record: INCOMPLETE, red', () => {
const r = run(['passed', 'missing', 'passed']);
expect(r.status).toBe(1);
expect(r.out).toContain(`PANEL ${ID} INCOMPLETE`);
});
test('a trial the harness never started: red, machine-classified for one re-dispatch', () => {
const r = run(['passed', 'harness', 'passed']);
expect(r.status).toBe(1);
expect(r.out).toContain('no trial record (never started)');
expect(r.out).toContain('INFRA-ONLY RED');
});
test('a contract trial at 2/3: red', () => {
const r = run(['passed', 'passed', 'contract']);
expect(r.status).toBe(1);
expect(r.out).toContain(`PANEL ${ID} FAIL 2/3`);
expect(r.out).toContain('contract violation');
expect(r.out).not.toContain('INFRA-ONLY RED');
});
test('quarantined 1/3: reported, does not fail the lane', () => {
const r = run(['passed', 'failed', 'failed'], true);
expect(r.status, r.out).toBe(0);
expect(r.out).toContain(`◌ ${ID} behavior (quarantined) FAIL 1/3`);
});
test('quarantined 0/3: hard break, red', () => {
const r = run(['failed', 'failed', 'failed'], true);
expect(r.status).toBe(1);
expect(r.out).toContain('quarantined hard break');
});
test('quarantined contract violation: red', () => {
const r = run(['passed', 'passed', 'contract'], true);
expect(r.status).toBe(1);
});
test('a missing trial slice: red', () => {
const r = run(['passed', 'passed', 'passed'], false, 2);
expect(r.status).toBe(1);
expect(r.out).toContain('slice 2/4 reported NO result');
expect(r.out).toContain(`PANEL ${ID} INCOMPLETE`);
});
test('a later run attempt never replaces the first attempt verdict', () => {
const r = run(['passed', 'failed', 'failed']);
expect(r.status).toBe(1);
const retry = slice(3, 4, [trialOutcome(3, 'passed')!]);
fs.mkdirSync(path.join(r.dir, 'paid-slice-3-a2'), { recursive: true });
fs.writeFileSync(path.join(r.dir, 'paid-slice-3-a2', 'slice-3.json'), JSON.stringify({ ...retry, attempt: 2 }));
const again = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--tier', 'periodic', '--report', r.dir],
{ cwd: ROOT, encoding: 'utf8', timeout: 30_000 });
expect(again.status).toBe(1);
expect(again.stdout).toContain('attempt 1 (later attempts 2 reported, never replacing it)');
});
test('verdicts become next-run receipts: a whole fresh PASS panel, and negatives for FAIL panels', () => {
const inputKey = 'b'.repeat(64);
const withKey = (results: TrialResult[]) => [1, 2, 3, 4].map(index => slice(index, 4, index === 4 ? [passed(RULE_A)]
: [{ ...trialOutcome(index, results[index - 1]!)!, inputKey }]));
const env = { GITHUB_RUN_ID: '77', GITHUB_SHA: 'e'.repeat(40) };
const green = report(plan(), withKey(['passed', 'failed', 'passed']), {}, env);
expect(green.status, green.out).toBe(0);
const receipt = JSON.parse(fs.readFileSync(path.join(green.dir, 'report-receipts', `${inputKey}.panel.json`), 'utf8'));
expect(receipt).toMatchObject({ key: inputKey, case: ID, panel: { n: 3, k: 2 }, source: { runId: '77/1' } });
expect(receipt.trials.map((t: any) => t.outcome)).toEqual(['passed', 'failed', 'passed']);
const red = report(plan(), withKey(['passed', 'failed', 'failed']), {}, env);
expect(red.status).toBe(1);
expect(fs.readdirSync(path.join(red.dir, 'report-receipts'))).toEqual([`${inputKey}.fail.json`]);
});
});
+30 -46
View File
@@ -7,7 +7,7 @@ import {
shardFile, sliceExecutionOrder, sliceSupervisedWallMs, CASE_SHARDED_FILES,
} from '../scripts/test-paid-shards';
import {
ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, RETRY_MAX_CASE_MS, SHORT_CASE_RETRY_FILES,
ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS,
FINDING_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS,
} from './helpers/eval-budgets';
@@ -15,16 +15,16 @@ import { E2E_TOUCHFILES } from './helpers/touchfiles';
const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8');
const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file));
// Walls cover every attempt the retry rule allows: files with a case budget
// past RETRY_MAX_CASE_MS run once (a timed-out attempt is a verdict).
// Paid evals never retry (approved 2026-09-29): each wall covers one run of
// every case plus the supervision reserve.
const expectedWalls = {
'test/skill-e2e-qa-callers.test.ts': 3_270_000,
'test/skill-e2e-qa-callers.test.ts': 1_695_000,
'test/skill-e2e-shared-libs-paths.test.ts': 1_920_000,
'test/skill-e2e-ship-docsync.test.ts': 4_920_000,
'test/skill-llm-eval.test.ts': 6_220_000,
'test/skill-llm-eval.test.ts': 3_170_000,
'test/skill-e2e-auq-consistency.test.ts': 1_080_000,
'test/skill-e2e-auq-matrix.test.ts': 3_720_000,
'test/skill-e2e-plan-format.test.ts': 2_600_000,
'test/skill-e2e-auq-matrix.test.ts': 1_920_000,
'test/skill-e2e-plan-format.test.ts': 1_360_000,
'test/skill-e2e-auto-decide-preserved.test.ts': 1_020_000,
'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_020_000,
'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_020_000,
@@ -33,38 +33,22 @@ const expectedWalls = {
'test/skill-e2e-plan-mode-no-op.test.ts': 3_120_000,
'test/skill-e2e-plan-ceo-mode-routing.test.ts': 1_320_000,
'test/skill-e2e-plan-eng-plan-mode.test.ts': 1_320_000,
'test/skill-e2e-plan-prosons.test.ts': 2_600_000,
'test/skill-e2e-plan-prosons.test.ts': 1_360_000,
'test/skill-e2e-plan.test.ts': 3_720_000,
};
test('retry rule: only files whose every case is CAPTURE tier or shorter retry; longer cases run once', () => {
expect(RETRY_MAX_CASE_MS).toBe(ALL_TIERS.CAPTURE_MS + 15_000);
test('paid evals never retry: every paid file and registered row runs once', () => {
for (const row of FILE_RETRY_BUDGETS) {
expect(row.retries, row.file).toBe(row.caseMs <= RETRY_MAX_CASE_MS ? (row.file.endsWith('plan-mode-no-op.test.ts') ? 2 : 1) : 0);
expect(retriesForFiles([row.file])).toBe(row.retries);
expect(Object.hasOwn(row, 'retries'), row.file).toBe(false);
expect(retriesForFiles([row.file])).toBe(0);
}
expect(FILE_RETRY_BUDGETS.filter(row => row.retries > 0).map(row => row.file).sort()).toEqual([
'test/skill-e2e-auq-matrix.test.ts', 'test/skill-e2e-plan-format.test.ts', 'test/skill-e2e-plan-prosons.test.ts',
'test/skill-e2e-qa-callers.test.ts', 'test/skill-llm-eval.test.ts',
]);
const paid = collectPaidTestFiles();
for (const file of SHORT_CASE_RETRY_FILES) {
expect(paid, `stale SHORT_CASE_RETRY_FILES entry: ${file}`).toContain(file);
expect(FILE_RETRY_BUDGETS.some(row => row.file === file)).toBe(false);
const source = read(file);
// Declared short budgets only: a JUDGE/CAPTURE tier or a literal at most the
// cap, no longer tier and no ms literal past the cap.
const literals = [...source.matchAll(/(?<![\w.])(\d{1,3}(?:_\d{3})+|\d{5,})(?![\w.])/g)]
.map(match => Number(match[1]!.replace(/_/g, '')));
expect(/\b(?:JUDGE_MS|CAPTURE_MS)\b/.test(source) || literals.some(ms => ms >= 60_000 && ms <= RETRY_MAX_CASE_MS), file).toBe(true);
expect(source, file).not.toMatch(/\b(?:CAPTURE_LONG_MS|PTY_MS|PTY_LONG_MS|OVERLAY_CASE_[A-Z_]+)\b/);
expect(literals.filter(ms => ms > RETRY_MAX_CASE_MS && ms < 10_000_000), file).toEqual([]);
expect(retriesForFiles([file])).toBe(1);
for (const file of collectPaidTestFiles()) expect(retriesForFiles([file]), file).toBe(0);
expect(buildPaidShardArgs(['test/x.test.ts'], 1000, 2)).toContain('--retry');
expect(buildPaidShardArgs(['test/x.test.ts'], 1000, 2).join(' ')).toContain('--retry 0');
const scripts: Record<string, string> = JSON.parse(read('package.json')).scripts;
for (const [name, command] of Object.entries(scripts)) {
if (/^test:(?:evals|e2e|gate|periodic)/.test(name)) expect(command, name).not.toMatch(/--retry(?:\s+|=)[1-9]/);
}
for (const file of paid.filter(file => !SHORT_CASE_RETRY_FILES.includes(file) && !FILE_RETRY_BUDGETS.some(row => row.file === file))) {
expect(retriesForFiles([file]), file).toBe(0);
}
expect(retriesForFiles([SHORT_CASE_RETRY_FILES[0]!, 'test/skill-e2e-plan.test.ts'])).toBe(0);
});
test('registration covers exactly the seventeen demonstrated full-file retry gaps', () => {
@@ -133,14 +117,14 @@ for (const row of newBudgets) {
outcomes: [{ files: [key], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count,
skippedTests: 0, budget: resolvePaidShardBudget([key]) }] }];
test(`${row.file}: full wall and existing retries propagate through planning`, () => {
expect(retriesForFiles([row.file])).toBe(row.retries);
test(`${row.file}: full wall propagates through planning and runs once`, () => {
expect(retriesForFiles([row.file])).toBe(0);
expect(resolvePaidShardBudget([row.file])).toEqual({ timeoutMs: expectedWalls[row.file as keyof typeof expectedWalls], source: 'registered', policyId: row.id });
expect(planPaidShards(['test/a.test.ts', row.file, 'test/z.test.ts'], { maxFilesPerShard: 3 })).toContainEqual([row.file]);
expect(() => resolvePaidShardBudget([row.file, 'test/neighbor.test.ts'])).toThrow('own shard');
expect(resolvePaidShardBudget([row.file], 50)).toEqual({ timeoutMs: 50, source: 'explicit', policyId: row.id });
expect(buildPaidShardArgs([row.file], row.shardMs, 2, retriesForFiles([row.file]))).toEqual([
'test', row.file, '--retry', String(row.retries), '--concurrent', '--max-concurrency=2', `--timeout=${row.shardMs}`,
'test', row.file, '--retry', '0', '--concurrent', '--max-concurrency=2', `--timeout=${row.shardMs}`,
]);
});
@@ -191,17 +175,17 @@ test('quality judge supervision includes the added judge without changing ordina
expect(ALL_TIERS).toEqual({ JUDGE_MS: 120000, CAPTURE_MS: 300000, CAPTURE_LONG_MS: 600000, PTY_MS: 900000, PTY_LONG_MS: 1200000 });
const quality = 'test/skill-llm-eval.test.ts';
const qualityBudget = FILE_RETRY_BUDGETS.find(row => row.file === quality)!;
expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 6_220_000, source: 'registered', policyId: qualityBudget.id });
expect(retriesForFiles([quality])).toBe(1);
expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 3_170_000, source: 'registered', policyId: qualityBudget.id });
expect(retriesForFiles([quality])).toBe(0);
const qualitySource = read(quality);
const judgeTimeouts = [...qualitySource.matchAll(/}\s*,\s*(JUDGE_MS|WORKFLOW_JUDGE_TEST_MS)\s*\);/g)].map(match => match[1]);
expect(judgeTimeouts.filter(timeout => timeout === 'JUDGE_MS')).toHaveLength(7);
expect(judgeTimeouts.filter(timeout => timeout === 'WORKFLOW_JUDGE_TEST_MS')).toHaveLength(17);
expect(qualitySource).toContain('WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000');
expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS');
expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000);
expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([
...Array(2).fill([1, 1500000, 0, 1620000]),
expect(qualityBudget.shardMs).toBe(7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000) + 120_000);
expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.shardMs])).toEqual([
...Array(2).fill([1, 1500000, 1620000]),
]);
for (const tier of ['gate', 'periodic'] as const) {
const m = buildRunManifest({ tier, sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } });
@@ -227,7 +211,7 @@ test('detached PR fallback and release commands cover their actual default worke
const prFloor = Math.ceil((Math.ceil(fullGateFiles.length / prWorkers) * 1_800_000 + fullGateFiles.reduce(
(total, file) => total + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - 1_800_000), 0,
)) / 1000 * 1.05);
expect(prFloor).toBe(77_501);
expect(prFloor).toBe(72_755);
expect(prWall).toBe(92_820_000);
expect(prWall).toBeGreaterThanOrEqual(paidShardWallUpperBoundMs(files, prWorkers) + 120_000);
@@ -246,8 +230,8 @@ test('detached PR fallback and release commands cover their actual default worke
)) / 1000 * 1.05));
}
const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000;
expect(releaseFloors).toEqual([26_471, 30_797]);
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(57_268);
expect(releaseFloors).toEqual([21_725, 33_821]);
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(55_546);
expect(detachedReleaseWall).toBe(116_700_000);
expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000);
});
@@ -301,7 +285,7 @@ test('both gate executors plan the complete census and supervise every planned s
}
});
test('the periodic executor supervises every actual case and retry within its planned CI wall', () => {
test('the periodic executor supervises every actual case within its planned CI wall', () => {
const workflow: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml'));
const executor = workflow.jobs['eval-slices'];
const emit = workflow.jobs['plan-slices'].steps.filter((step: any) =>
@@ -318,7 +302,7 @@ test('the periodic executor supervises every actual case and retry within its pl
evalsAll: true, env: { EVALS_ALL: '1' } });
const census = manifest.entries.filter(row => row.status === 'planned');
expect(new Set(census.map(row => shardFile(row.file)))).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected));
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_220_000);
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(3_170_000);
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
census.filter(row => row.slice === i + 1)).map(row => row.file), active.jobs));
expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
+120 -7
View File
@@ -37,11 +37,17 @@ import {
summarize,
summaryExitCode,
verifySliceResults,
expandTrialShards,
formatCapacityPreflight,
panelReports,
shardSlug,
type PaidRunManifest,
type ShardOutcome,
type SliceResult,
} from '../scripts/test-paid-shards';
import { E2E_KINDS } from './helpers/touchfiles-data';
const ROOT = path.resolve(__dirname, '..');
const outcome = (over: Partial<ShardOutcome>): ShardOutcome => ({
@@ -487,8 +493,8 @@ describe('hollow-shard guard', () => {
});
describe('retry parity', () => {
test('registered native workflows follow the retry rule while overlay attempts stay isolated', () => {
// A 25-minute case is past RETRY_MAX_CASE_MS: its timed-out attempt is the verdict.
test('registered native workflows and overlays run once', () => {
// Paid evals never retry: a timed-out attempt is the verdict.
const native = 'test/skill-e2e-plan-ceo-split-overflow.test.ts';
expect(retriesForFiles([native])).toBe(0);
expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(0);
@@ -496,16 +502,123 @@ describe('retry parity', () => {
const overlay = 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts';
expect(retriesForFiles([overlay])).toBe(0);
});
test('the matrix-era earned retries now follow the timeout-is-a-verdict rule, and each names a real file', () => {
// These three old matrix rows earned `retries: 2`; every one has a
// CAPTURE_LONG case, so a timed-out attempt is now their verdict.
test('the matrix-era earned retries are retired, and each names a real file', () => {
// These three old matrix rows earned `retries: 2`; paid evals never retry.
for (const file of ['test/skill-e2e-office-hours-auto-mode.test.ts', 'test/skill-e2e-plan-mode-no-op.test.ts', 'test/skill-e2e-workflow.test.ts']) {
expect(fs.existsSync(path.join(ROOT, file)), `stale retry parity entry: ${file}`).toBe(true);
expect(retriesForFiles([file])).toBe(0);
}
expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(0);
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1);
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(0);
expect(buildPaidShardArgs(['x'], 1000, 4, 2)).toContain('2');
expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 1');
expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 0');
});
});
describe('trial planner (behavior and quarantined panels)', () => {
const REVIEW = 'test/skill-e2e-review.test.ts';
const budgetPlan = (tier: 'gate' | 'periodic', kinds: Record<string, 'rule' | 'behavior' | 'judge'>, quarantine: Record<string, unknown> = {}) =>
buildRunManifest({ tier, sliceBudgetMs: 540_000, jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' },
kinds: { ...E2E_KINDS, ...kinds }, quarantine });
test('a behavior case becomes three trial shards on three different slices; its file shard runs the rest', () => {
const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' });
const trials = manifest.entries.filter(entry => entry.file.startsWith(`${REVIEW}#review-sql-injection~t`));
expect(trials.map(entry => entry.file)).toEqual([1, 2, 3].map(n => `${REVIEW}#review-sql-injection~t${n}`));
expect(trials.every(entry => entry.status === 'planned')).toBe(true);
expect(new Set(trials.map(entry => entry.slice)).size).toBe(3);
expect(trials[0]!.trial).toEqual({ kind: 'behavior', panel: { n: 3, k: 2 }, quarantined: false });
const fileShard = manifest.entries.find(entry => entry.file === REVIEW)!;
expect(fileShard.excludeCases).toEqual(['review-sql-injection']);
const slugs = manifest.entries.map(entry => shardSlug([entry.file]));
expect(new Set(slugs).size).toBe(slugs.length);
expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest);
});
test('a file whose only tier case is isolated drops its file shard', () => {
const manifest = budgetPlan('periodic', { 'review-design-lite': 'behavior' });
expect(manifest.entries.some(entry => entry.file === REVIEW)).toBe(false);
expect(manifest.entries.filter(entry => entry.file.startsWith(`${REVIEW}#`)).map(entry => entry.file))
.toEqual([1, 2, 3].map(n => `${REVIEW}#review-design-lite~t${n}`));
});
test('a quarantined rule case runs a full panel with k = n', () => {
const manifest = budgetPlan('gate', {}, { 'review-enum-completeness': { reason: 'r' } });
const trial = manifest.entries.find(entry => entry.file === `${REVIEW}#review-enum-completeness~t1`)!;
expect(trial.trial).toEqual({ kind: 'rule', panel: { n: 3, k: 3 }, quarantined: true });
});
test('slice-count plans keep trials on different slices too', () => {
const manifest = buildRunManifest({ tier: 'gate', sliceCount: 5, evalsAll: true, env: { EVALS_ALL: '1' },
kinds: { ...E2E_KINDS, 'review-sql-injection': 'behavior', 'review-enum-completeness': 'behavior' } });
for (const id of ['review-sql-injection', 'review-enum-completeness']) {
const slices = manifest.entries.filter(entry => entry.file.startsWith(`${REVIEW}#${id}~t`)).map(entry => entry.slice);
expect(new Set(slices).size).toBe(3);
}
});
test('judges and unknown ids cannot be isolated; unknown registrations throw', () => {
expect(() => budgetPlan('gate', { 'review/SKILL.md workflow': 'behavior' })).toThrow(/Only live E2E cases/);
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'trial-unknown-'));
try {
fs.mkdirSync(path.join(dir, 'test'));
fs.writeFileSync(path.join(dir, 'test/skill-e2e-x.test.ts'), 'const name = pick(); runSkillTest({ testName: name });');
expect(() => expandTrialShards(['test/skill-e2e-x.test.ts'], 'gate', dir, {
kinds: { x: 'behavior' }, touchfiles: { x: ['test/skill-e2e-x.test.ts'] }, tiers: { x: 'gate' },
})).toThrow(/statically known case registration/);
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
});
test('parse rejects partial panels, shared runners, forged plans and stray exclusions', () => {
const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' });
const trialFiles = [1, 2, 3].map(n => `${REVIEW}#review-sql-injection~t${n}`);
const mutate = (fn: (m: PaidRunManifest) => void) => { const m = structuredClone(manifest); fn(m); return JSON.stringify(m); };
expect(() => parseRunManifest(mutate(m => { m.entries = m.entries.filter(e => e.file !== trialFiles[1]); })))
.toThrow(/exactly its 3 trials/);
expect(() => parseRunManifest(mutate(m => {
const [a, b] = trialFiles.map(f => m.entries.find(e => e.file === f)!);
b!.slice = a!.slice;
}))).toThrow(/share a slice/);
expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === trialFiles[0])!.trial!.panel.k = 1; })))
.toThrow(/fixed policy panel/);
expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === trialFiles[0])!.trial = undefined; })))
.toThrow(/fixed policy panel/);
expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === REVIEW)!.excludeCases = ['review-design-lite']; })))
.toThrow(/exclude only cases/);
expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === REVIEW)!.trial = m.entries.find(e => e.file === trialFiles[0])!.trial; })))
.toThrow(/Only trial shards/);
});
test('capacity preflight names slices, shards, waves and the longest indivisible trial', () => {
const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' });
const lines = formatCapacityPreflight(manifest, 16).join('\n');
expect(lines).toContain(`${manifest.sliceCount} slice(s)`);
expect(lines).toContain('3 trial shard(s)');
expect(lines).toContain(`wave(s) at max-parallel 16: ${Math.ceil(manifest.sliceCount / 16)}`);
expect(lines).toMatch(/longest indivisible trial ~\d+\.\dm \(test\/skill-e2e-review\.test\.ts#review-sql-injection~t\d\)/);
});
test('durations: trials record their longest wall under the case key and seed their own estimate', () => {
const key = `${REVIEW}#review-sql-injection`;
const merged = mergePaidTestDurations({}, [{ version: 1, tier: 'gate', sliceIndex: 1, sliceCount: 1, outcomes: [1, 2, 3].map(n => ({
files: [`${key}~t${n}`], status: 'passed' as const, exitCode: 0, elapsedMs: n * 60_000, executedTests: 1, skippedTests: 0,
})) }]);
expect(merged).toEqual({ [key]: 180_000 });
const packed = packBySliceBudget([1, 2, 3].map(n => `${key}~t${n}`), 540_000, 2, merged);
expect(packed.slices).toHaveLength(3);
expect(Object.values(packed.estimates)).toEqual([180_000, 180_000, 180_000]);
});
test('reuse is whole-panel only: a panel mixing reused and fresh trials is INCOMPLETE', () => {
const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' });
const trials = manifest.entries.filter(entry => entry.trial);
const reused = { inputKey: 'c'.repeat(64), runId: '1001/1', revision: 'd'.repeat(40), completedAt: 1 };
const results = (reusedTrials: number[]): SliceResult[] => trials.map(entry => ({ version: 1, tier: 'gate', sliceIndex: entry.slice,
sliceCount: manifest.sliceCount, outcomes: [{ files: [entry.file], status: 'passed', exitCode: 0, elapsedMs: 1, executedTests: 1, skippedTests: 0,
trial: { case: 'review-sql-injection', trial: Number(entry.file.slice(-1)), ...entry.trial!, outcome: 'passed', cost_usd: 0, duration_ms: 1 },
...(reusedTrials.includes(Number(entry.file.slice(-1))) ? { reused } : {}) }] }));
expect(panelReports(manifest, results([]), 1)[0]).toMatchObject({ status: 'PASS' });
expect(panelReports(manifest, results([1, 2, 3]), 1)[0]).toMatchObject({ status: 'PASS' });
expect(panelReports(manifest, results([2]), 1)[0]).toMatchObject({ status: 'INCOMPLETE', failsLane: true, reason: expect.stringContaining('partial panel reuse') });
});
});
+122
View File
@@ -47,6 +47,17 @@ import {
selectPaidTestFiles,
buildRunManifest,
parseRunManifest,
classifyTrialShard,
sliceExitCode,
guardTrialRecords,
parseJUnitCases,
caseIdForTestName,
shardTrial,
excludedCasesNamePattern,
runCaseDiagnosis,
caseFile,
parseCliOptions,
type CaseTrialPlan,
type ShardOutcome,
} from '../scripts/test-paid-shards';
@@ -578,3 +589,114 @@ describe('all-skipped pass census', () => {
expect(reviewLine).not.toContain('SKIPPED');
});
});
describe('isolated trial shards: record, classification and slice exit', () => {
const plan: CaseTrialPlan = { kind: 'behavior', panel: { n: 3, k: 2 }, quarantined: false };
const key = (n: number) => `test/skill-e2e-review.test.ts#review-sql-injection~t${n}`;
const base = { status: 'passed' as const, executedTests: 1, skippedTests: 0, elapsedMs: 5 };
const none = { records: [], contract: null };
test('trial keys keep their case id, file and index', () => {
expect(shardCaseId(key(2))).toBe('review-sql-injection');
expect(shardFile(key(2))).toBe('test/skill-e2e-review.test.ts');
expect(shardTrial(key(2))).toBe(2);
expect(shardTrial('test/skill-e2e-review.test.ts#review-sql-injection')).toBeNull();
expect(shardSlug([key(2)])).toBe('skill-e2e-review--review-sql-injection.t2');
});
test('classification: verdicts versus harness problems', () => {
const c = (over: Partial<ShardOutcome>, evidence: { records: any[]; contract: string | null } = none) =>
classifyTrialShard({ ...base, ...over }, 'review-sql-injection', 1, plan, evidence);
expect(c({}).outcome).toBe('passed');
expect(c({}, { records: [], contract: 'handoff missing' })).toMatchObject({ outcome: 'failed', failure_class: 'contract', error: 'handoff missing' });
expect(c({ status: 'failed' }, { records: [{ passed: false, exit_reason: 'timeout', timeout_at_turn: 9, error: 'x' }], contract: null }))
.toMatchObject({ outcome: 'failed', failure_class: 'timeout', timeout_at_turn: 9 });
expect(c({ status: 'failed' }, { records: [{ passed: false, error: 'expected 3' }], contract: null })).toMatchObject({ outcome: 'failed', failure_class: 'assertion' });
expect(c({ status: 'timed-out', executedTests: null, skippedTests: null })).toMatchObject({ outcome: 'failed', failure_class: 'timeout' });
expect(c({ status: 'failed', executedTests: null, skippedTests: null })).toMatchObject({ outcome: 'failed', failure_class: 'infra' });
expect(c({ status: 'failed', executedTests: 0, skippedTests: 0 })).toMatchObject({ outcome: 'failed', failure_class: 'infra' });
expect(c({ executedTests: 1, skippedTests: 1 })).toMatchObject({ outcome: 'skipped' });
for (const over of [{ status: 'never-started' as const }, { status: 'passed-empty' as const }, { executedTests: 2 },
{ runnerError: 'spawn failed' }, { executedTests: 0, skippedTests: 0 }]) {
expect(c(over).outcome, JSON.stringify(over)).toBeNull();
}
});
test('slice exit: rule shards stay strict; failed trials never red the runner, missing records do', () => {
const trial = (outcome: 'passed' | 'failed' | null) => ({ status: outcome === 'failed' ? 'failed' as const : 'passed' as const,
trial: { case: 'c', trial: 1, ...plan, outcome, cost_usd: 0, duration_ms: 1, ...(outcome === null ? { harness: 'never started' } : {}) } });
expect(sliceExitCode([{ status: 'passed' }, trial('failed')])).toBe(0);
expect(sliceExitCode([{ status: 'failed' }, trial('passed')])).toBe(1);
expect(sliceExitCode([{ status: 'passed' }, trial(null)])).toBe(1);
expect(sliceExitCode([{ status: 'timed-out' }])).toBe(1);
const hollow = guardTrialRecords([{ ...trial('passed'), status: 'passed-empty' as const }]);
expect(hollow[0]!.trial!.outcome).toBeNull();
expect(sliceExitCode(hollow)).toBe(1);
});
test('runPaidShards binds each trial to its case, index and panel and records its outcome', async () => {
const evalDirBase = fs.mkdtempSync(path.join(os.tmpdir(), 'trial-shards-'));
try {
const script = (fail: boolean) => `const fs = require('fs'), path = require('path');
const dir = process.env.GSTACK_EVAL_DIR; fs.mkdirSync(dir, { recursive: true });
const env = Object.fromEntries(Object.entries(process.env).filter(([k]) => k.startsWith('GSTACK_EVAL_') || k === 'EVALS_SELECTION_JSON'));
fs.writeFileSync(path.join(dir, 'env.json'), JSON.stringify(env));
fs.writeFileSync(path.join(dir, 'run.json'), JSON.stringify({ tests: [{ name: 'review-sql-injection', passed: ${!fail}, cost_usd: 0.5,
duration_ms: 1, exit_reason: ${fail ? "'timeout'" : "'success'"}, timeout_at_turn: 4, model: 'm' }] }));
console.log("Ran 1 tests across 1 files. [1ms]"); process.exit(${fail ? 1 : 0});`;
const summary = await runPaidShards([[key(1)], [key(2)], [key(3)]], {
jobs: 3, evalDirBase, log: () => {}, trials: { [key(1)]: plan, [key(2)]: plan, [key(3)]: plan },
commandFor: files => ({ command: process.execPath, args: ['-e', script(files[0] === key(2))] }),
});
const byKey = (n: number) => summary.outcomes.find(o => o.files[0] === key(n))!;
expect(byKey(1).trial).toMatchObject({ case: 'review-sql-injection', trial: 1, outcome: 'passed', cost_usd: 0.5, model: 'm' });
expect(byKey(2).trial).toMatchObject({ trial: 2, outcome: 'failed', failure_class: 'timeout', exit_reason: 'timeout', timeout_at_turn: 4 });
expect(sliceExitCode(summary.outcomes)).toBe(0);
const env = JSON.parse(fs.readFileSync(path.join(evalDirBase, 'shards', shardSlug([key(3)]), 'env.json'), 'utf8'));
expect(env).toMatchObject({ GSTACK_EVAL_CASE_ID: 'review-sql-injection', GSTACK_EVAL_KIND: 'behavior', GSTACK_EVAL_TRIAL: '3',
GSTACK_EVAL_PANEL_N: '3', GSTACK_EVAL_PANEL_K: '2', GSTACK_EVAL_POLICY_VERSION: '1' });
expect(JSON.parse(env.EVALS_SELECTION_JSON).selected).toEqual(['review-sql-injection']);
} finally { fs.rmSync(evalDirBase, { recursive: true, force: true }); }
});
test('file shards exclude isolated names; JUnit cases map to registry ids or stay unattributed', () => {
const pattern = new RegExp(excludedCasesNamePattern(['review-sql-injection']));
expect(pattern.test('suite > review-sql-injection')).toBe(false);
expect(pattern.test('suite > review-enum-completeness')).toBe(true);
const cases = parseJUnitCases(`<testsuites><testsuite name="f">
<testcase name="review-sql-injection" classname="s" time="1.5" />
<testcase name="review-enum-completeness" classname="s" time="0.1"><failure type="TimeoutError" message="test &amp; timed out" /></testcase>
<testcase name="plain helper" classname="" time="0"><skipped /></testcase>
</testsuite></testsuites>`);
expect(cases).toEqual([
{ name: 'review-sql-injection', classname: 's', outcome: 'passed', timeMs: 1500 },
{ name: 'review-enum-completeness', classname: 's', outcome: 'failed', timeMs: 100, failureType: 'TimeoutError', message: 'test & timed out' },
{ name: 'plain helper', classname: '', outcome: 'skipped', timeMs: 0 },
]);
expect(caseIdForTestName('review-sql-injection')).toBe('review-sql-injection');
expect(caseIdForTestName(CASE_TEST_NAMES['plan-review-report']!)).toBe('plan-review-report');
expect(caseIdForTestName('plain helper')).toBeNull();
});
test('--case/--trials: local diagnosis flags are validated and never combine with CI modes', () => {
expect(parseCliOptions(['--case', 'review-sql-injection', '--trials', '5'], {})).toMatchObject({ caseId: 'review-sql-injection', trials: 5 });
expect(() => parseCliOptions(['--trials', '3'], {})).toThrow('--trials requires --case');
expect(() => parseCliOptions(['--case', 'no-such-case'], {})).toThrow('live E2E case id');
expect(() => parseCliOptions(['--case', 'review-sql-injection', '--report', '/tmp/r'], {})).toThrow('local diagnosis');
expect(caseFile('review-sql-injection')).toBe('test/skill-e2e-review.test.ts');
});
test('--case runs the CI panel runner and prints its panelVerdict', async () => {
const evalDirBase = fs.mkdtempSync(path.join(os.tmpdir(), 'case-diagnosis-'));
const lines: string[] = [];
try {
const verdict = await runCaseDiagnosis('review-sql-injection', { trials: 3, evalDirBase, log: line => lines.push(line), jobs: 3,
commandFor: files => ({ command: process.execPath, args: ['-e',
`console.log("Ran 1 tests across 1 files. [1ms]"); process.exit(${files[0]!.endsWith('~t3') ? 1 : 0});`] }) });
// A rule case keeps its meaning locally: every trial must pass.
expect(verdict).toMatchObject({ case: 'review-sql-injection', kind: 'rule', panel: { n: 3, k: 3 }, passed: 2, status: 'FAIL' });
expect(lines.join('\n')).toContain('--case review-sql-injection: 3 trial(s) of test/skill-e2e-review.test.ts');
expect(lines.join('\n')).toContain('FAIL 2/3 (✓✓✗)');
} finally { fs.rmSync(evalDirBase, { recursive: true, force: true }); }
});
});
+30 -2
View File
@@ -9,10 +9,11 @@ import { describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { CASE_CI_EXCLUDE, PERIODIC_CI_EXCLUDE } from './helpers/periodic-exclude-data';
import { CASE_CI_EXCLUDE, CASE_QUARANTINE, EVAL_POLICY, PERIODIC_CI_EXCLUDE } from './helpers/periodic-exclude-data';
import { E2E_TOUCHFILES } from './helpers/touchfiles';
import { quarantinePolicyProblems } from '../scripts/eval-flake-rank';
import { isPaidTestFile } from './helpers/paid-test-set';
import { buildRunManifest, CASE_SHARDED_FILES, expandCaseShards, partitionCaseExclusions, selectPaidTestFiles, shardCaseId, shardFile } from '../scripts/test-paid-shards';
import { buildRunManifest, CASE_SHARDED_FILES, expandCaseShards, fileCaseRegistration, partitionCaseExclusions, selectPaidTestFiles, shardCaseId, shardFile } from '../scripts/test-paid-shards';
const ROOT = path.resolve(__dirname, '..');
@@ -72,3 +73,30 @@ describe('periodic exclude policy', () => {
}
});
});
describe('eval verdict policy (pre-registered)', () => {
test('EVAL_POLICY carries exactly the approved constants; a change needs re-approval and a version bump', () => {
expect(EVAL_POLICY).toEqual({
version: 1,
panel: { n: 3, k: 2 },
quarantine: { entry: { rate: 0.95, minTrials: 10 }, exit: { rate: 0.97, minTrials: 10 }, capFraction: 0.10, expiryWeeklyRuns: 8 },
judge: { samples: 3 },
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
infraRedispatch: 1,
});
});
test('every CASE_QUARANTINE entry is a diagnosed, dated, non-product blocking case within the tier cap', () => {
expect(quarantinePolicyProblems(CASE_QUARANTINE).map(problem => problem.message)).toEqual([]);
});
test('a quarantined case runs as isolated trial shards: its files register it literally', () => {
for (const id of Object.keys(CASE_QUARANTINE)) {
const files = (E2E_TOUCHFILES[id] ?? []).filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file));
expect(files.length, `${id}: no paid file registers it`).toBeGreaterThan(0);
for (const file of files) {
expect(fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')).known, `${id}: ${file} registration must be statically known`).toBe(true);
}
}
});
});
+9 -11
View File
@@ -1,4 +1,4 @@
/** The real review registrations must finish capture cleanup before Bun retries. */
/** The real review registrations record late results and clean up before finalization, under the production zero-retry arguments. */
import { expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as os from 'node:os';
@@ -12,7 +12,7 @@ const CASES = [
['review-design-lite', 400, 35],
] as const;
for (const [id, workMs, maxTurns] of CASES) {
test.each(['recover', 'both-timeout'])(`${id} records late results before retry or finalization: %s`, scenario => {
test.each(['success', 'timeout'])(`${id} records late results before finalization: %s`, scenario => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'review-finalization-'));
const script = path.join(dir, 'registration.test.ts');
const facts = path.join(dir, 'events.jsonl');
@@ -41,7 +41,7 @@ mock.module(path.join(root, 'test/helpers/session-runner.ts'), () => ({
runSkillTest: async opts => {
const id = ++attempts;
event({ kind: 'start', id, timeout: opts.timeout, maxTurns: opts.maxTurns, cwd: opts.workingDirectory });
const timeout = id === 1 || ${JSON.stringify(scenario)} === 'both-timeout';
const timeout = ${JSON.stringify(scenario)} === 'timeout';
// Use the caller's actual work budget; only the provider and budget
// constants are scaled. The actual registered Bun outer deadline stays.
await new Promise(resolve => setTimeout(resolve, timeout ? opts.timeout + 50 : 80));
@@ -62,27 +62,25 @@ await import(path.join(root, ${JSON.stringify(PAID_FILE)}));
`);
try {
const retries = retriesForFiles([PAID_FILE]);
expect(retries).toBe(1);
expect(retries).toBe(0);
const child = Bun.spawnSync([process.execPath, ...buildPaidShardArgs([script], resolvePaidShardTimeoutMs([PAID_FILE]), 2, retries)], {
cwd: ROOT, timeout: 15_000, stdout: 'pipe', stderr: 'pipe',
env: { ...process.env, EVALS: '', EVALS_ALL: '', TMPDIR: dir, TMP: dir, TEMP: dir },
});
const output = child.stdout.toString() + child.stderr.toString();
expect(child.exitCode, output).toBe(scenario === 'recover' ? 0 : 1);
expect(child.exitCode, output).toBe(scenario === 'success' ? 0 : 1);
expect(output).not.toContain('Unhandled error between tests');
const events = fs.readFileSync(facts, 'utf8').trim().split('\n').map(line => JSON.parse(line));
const starts = events.filter(event => event.kind === 'start');
const ready = events.filter(event => event.kind === 'ready');
const records = events.filter(event => event.kind === 'record');
expect(starts.map(event => event.id)).toEqual([1, 2]);
expect(starts.map(event => event.id)).toEqual([1]);
expect(starts.map(({ timeout, maxTurns }) => ({ timeout, maxTurns })))
.toEqual([{ timeout: workMs, maxTurns }, { timeout: workMs, maxTurns }]);
expect(ready).toEqual([{ kind: 'ready', id: 1, fixtureExists: true }, { kind: 'ready', id: 2, fixtureExists: true }]);
.toEqual([{ timeout: workMs, maxTurns }]);
expect(ready).toEqual([{ kind: 'ready', id: 1, fixtureExists: true }]);
expect(records.map(event => [event.id, event.exitReason]))
.toEqual([[1, 'timeout'], [2, scenario === 'recover' ? 'success' : 'timeout']]);
.toEqual([[1, scenario === 'success' ? 'success' : 'timeout']]);
expect(events.findIndex(event => event.kind === 'record' && event.id === 1))
.toBeLessThan(events.findIndex(event => event.kind === 'start' && event.id === 2));
expect(events.findIndex(event => event.kind === 'record' && event.id === 2))
.toBeLessThan(events.findIndex(event => event.kind === 'finalized'));
expect(events.filter(event => event.kind === 'finalized')).toHaveLength(1);
expect(events.find(event => event.kind === 'registration')).toEqual({ kind: 'registration', name: id, outerMs: workMs + 50 + 5_000 });
+3 -3
View File
@@ -169,7 +169,7 @@ describe('setup-gbrain owned Path 4 fixture', () => {
pathToClaudeCodeExecutable: '/nonexistent/free-test-never-spawn-claude',
signal: controller.signal,
}, () => { validated = true; }, mode === 'deadline' ? 250 : 1000, {
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as EvalCollector,
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as unknown as EvalCollector,
name: 'setup-gbrain-path4-local-pglite', suite: 'setup-gbrain',
});
} catch (error) { failure = String(error); }
@@ -214,7 +214,7 @@ describe('setup-gbrain owned Path 4 fixture', () => {
await Promise.resolve();
controller.abort(new Error('caller cancelled during validation'));
}, 1000, {
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as EvalCollector,
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as unknown as EvalCollector,
name: 'setup-gbrain-path4-local-pglite', suite: 'setup-gbrain',
})).rejects.toThrow('caller cancelled during validation');
expect(rows).toHaveLength(1);
@@ -288,7 +288,7 @@ describe('setup-gbrain owned Path 4 fixture', () => {
throw new Error(`assertion diagnostic ${fixture.token} ${credentialUrl}`);
}
}, undefined, {
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as EvalCollector,
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as unknown as EvalCollector,
name: 'setup-gbrain-path4-local-pglite', suite: 'setup-gbrain',
});
} catch (error) { thrown = String(error); }
+2 -2
View File
@@ -13,9 +13,9 @@ import { DEFAULT_SHARD_TIMEOUT_MS, retriesForFiles } from '../scripts/test-paid-
const cases: ShipHookCase[] = ['ship-managed-hook-refresh', 'ship-unmanaged-hook-consent', 'ship-local-hook-preservation'];
type Fault = 'skip-guard' | 'skip-consent' | 'ask-overwrite' | 'direct-install' | 'read-receipts' | 'edit-policy' | 'tamper-receipts' | 'repeat-question' | 'rate-limit';
test('whole-file supervision covers every F5 case and the unchanged Bun retry', () => {
test('whole-file supervision covers every F5 case run once', () => {
for (const [file, count] of [['test/skill-e2e-ship-hook-refresh.test.ts', 1], ['test/skill-e2e-ship-hook-consent.test.ts', 2]] as const) {
expect(retriesForFiles([file])).toBe(1);
expect(retriesForFiles([file])).toBe(0);
expect(count * CAPTURE_MS * (retriesForFiles([file]) + 1) + 120000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS);
}
});
+2 -2
View File
@@ -152,9 +152,9 @@ function protocol(fault?: Fault, billing?: Array<number | undefined>, controls:
return { provider, directory: () => directory, calls: () => calls, sessions };
}
test('one bounded native case preserves the existing whole-file retry allowance', () => {
test('one bounded native case fits the whole-file wall and never retries', () => {
const retries = retriesForFiles(['test/skill-e2e-ship-skip.test.ts']);
expect(retries).toBe(1);
expect(retries).toBe(0);
expect(CAPTURE_MS * (retries + 1) + 120000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS);
});
+55 -46
View File
@@ -14,7 +14,7 @@ import { afterAll, expect } from 'bun:test';
import { JUDGE_MS } from './helpers/eval-budgets';
import * as fs from 'fs';
import * as path from 'path';
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelMajority, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS } from './helpers/llm-judge';
import { ENG_REVIEW_EXCERPT } from './helpers/workflow-excerpt';
import type { JudgeScore } from './helpers/llm-judge';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA, type WorkflowJudgeInput } from './helpers/workflow-judge-input';
@@ -100,8 +100,9 @@ describeIfSelected('LLM-as-judge quality evals', [
// rewrites the pin).
const section = sliceBrowseSection('## Snapshot Flags');
const scores = await judge('browse skill reference (flags + commands)', section);
console.log('Browse SKILL.md scores:', JSON.stringify(scores, null, 2));
const samples = await judgePanel(() => judge('browse skill reference (flags + commands)', section));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('Browse SKILL.md panel:', JSON.stringify({ mean: scores, samples }, null, 2));
const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json');
const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8'));
@@ -120,9 +121,9 @@ describeIfSelected('LLM-as-judge quality evals', [
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4 && regressions.length === 0,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: regressions.length ? `${scores.reasoning} | ${regressions.join('; ')}` : scores.reasoning,
judge_reasoning: regressions.length ? `${judgePanelReasoning(samples)} | ${regressions.join('; ')}` : judgePanelReasoning(samples),
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
@@ -144,8 +145,9 @@ describeIfSelected('LLM-as-judge quality evals', [
if (setupStart < 0 || setupEnd < 0) throw new Error('browse/SKILL.md: setup block not found — regenerate with: bun run gen:skill-docs');
const section = content.slice(setupStart, setupEnd);
const scores = await judge('setup/binary discovery instructions', section);
console.log('Setup block scores:', JSON.stringify(scores, null, 2));
const samples = await judgePanel(() => judge('setup/binary discovery instructions', section));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('Setup block panel:', JSON.stringify({ mean: scores, samples }, null, 2));
evalCollector?.addTest({
name: 'setup block',
@@ -153,9 +155,9 @@ describeIfSelected('LLM-as-judge quality evals', [
tier: 'llm-judge',
passed: scores.actionability >= 3 && scores.clarity >= 3,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
// Setup block is intentionally minimal (binary discovery only).
@@ -203,7 +205,7 @@ describeIfSelected('QA skill quality evals', ['qa/SKILL.md workflow', 'qa/SKILL.
startMarker: '# /qa: Test', endMarker: null,
references: ['qa/templates/functional-report-template.md'] }).text;
const scores = await callJudge<JudgeScore>(`You are evaluating the quality of a QA testing workflow document for an AI coding agent.
const samples = await judgePanel(() => callJudge<JudgeScore>(`You are evaluating the quality of a QA testing workflow document for an AI coding agent.
The agent reads this source-file bundle to select browser, native functional or mixed
surfaces, explore with bounded probes, reproduce and diagnose defects, add a regression
@@ -222,8 +224,9 @@ Respond with ONLY valid JSON:
Here is the QA workflow to evaluate:
${section}`);
console.log('QA workflow scores:', JSON.stringify(scores, null, 2));
${section}`));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('QA workflow panel:', JSON.stringify({ mean: scores, samples }, null, 2));
evalCollector?.addTest({
name: 'qa/SKILL.md workflow',
@@ -231,9 +234,9 @@ ${section}`);
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
@@ -247,7 +250,7 @@ ${section}`);
const t0 = Date.now();
const section = sliceQaPatterns('## Health Score Rubric');
const scores = await callJudge<JudgeScore>(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score.
const samples = await judgePanel(() => callJudge<JudgeScore>(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score.
The agent uses this rubric after QA testing a website. It needs to:
1. Understand each scoring category and what counts as a deduction
@@ -264,8 +267,9 @@ Respond with ONLY valid JSON:
Here is the rubric to evaluate:
${section}`);
console.log('QA health rubric scores:', JSON.stringify(scores, null, 2));
${section}`));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('QA health rubric panel:', JSON.stringify({ mean: scores, samples }, null, 2));
evalCollector?.addTest({
name: 'qa/SKILL.md health rubric',
@@ -273,9 +277,9 @@ ${section}`);
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
@@ -294,7 +298,7 @@ ${section}`);
const diffAwareSection = sliceQaPatterns('### Diff-aware', '### Full');
const rulesSection = sliceQaPatterns('## Important Rules');
const result = await callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario.
const samples = await judgePanel(() => callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario.
SCENARIO:
A user runs /qa (a browser-based QA testing skill). The branch diff shows ONLY prompt template files and config file changes — no routes, views, controllers, components, or CSS were changed. The changes are "purely backend" with no obvious UI surface.
@@ -318,9 +322,10 @@ Respond with ONLY valid JSON:
Rules:
- would_browse should be true if the document instructs the agent to always use the browser regardless of diff content
- would_browse should be false if the document allows the agent to skip browser testing for non-UI changes
- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`);
- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`));
const result = { would_browse: judgePanelMajority(samples, 'would_browse'), ...judgePanelMean(samples, ['confidence'] as const) };
console.log('QA anti-refusal result:', JSON.stringify(result, null, 2));
console.log('QA anti-refusal panel:', JSON.stringify({ result, samples }, null, 2));
evalCollector?.addTest({
name: 'qa/SKILL.md anti-refusal',
@@ -328,9 +333,9 @@ Rules:
tier: 'llm-judge',
passed: result.would_browse === true && result.confidence >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { would_browse: result.would_browse ? 1 : 0, confidence: result.confidence },
judge_reasoning: result.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(result.would_browse).toBe(true);
@@ -362,7 +367,7 @@ describeIfSelected('Cross-skill consistency evals', ['cross-skill greptile consi
extractGrepLines(retroContent, 'retro/SKILL.md'),
].join('\n\n');
const result = await callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently.
const samples = await judgePanel(() => callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently.
INTENDED ARCHITECTURE:
- greptile-history has TWO paths: per-project (~/.gstack/projects/{slug}/greptile-history.md) and global (~/.gstack/greptile-history.md)
@@ -383,9 +388,10 @@ Evaluate consistency. Respond with ONLY valid JSON:
"reasoning": "brief explanation"
}
score (1-5): 5 = perfectly consistent, 1 = contradictory`);
score (1-5): 5 = perfectly consistent, 1 = contradictory`));
const result = { consistent: judgePanelMajority(samples, 'consistent'), ...judgePanelMean(samples, ['score'] as const) };
console.log('Cross-skill consistency:', JSON.stringify(result, null, 2));
console.log('Cross-skill consistency panel:', JSON.stringify({ result, samples }, null, 2));
evalCollector?.addTest({
name: 'cross-skill greptile consistency',
@@ -393,9 +399,9 @@ score (1-5): 5 = perfectly consistent, 1 = contradictory`);
tier: 'llm-judge',
passed: result.consistent && result.score >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { consistency_score: result.score },
judge_reasoning: result.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(result.consistent).toBe(true);
@@ -439,7 +445,8 @@ async function runWorkflowJudge(opts: {
const workDeadline = started + JUDGE_MS;
let stage: 'input' | 'judge' | 'validation' | 'recording' = 'input';
let finalized = false;
let scores: JudgeScore | undefined;
let samples: JudgeScore[] | undefined;
let scores: Record<typeof JUDGE_SCORE_DIMENSIONS[number], number> | undefined;
let manualReview: ManualJudgeReview | undefined;
let customInputMetadata: { prompt: string; model: string } | undefined;
let reused: ReturnType<ReturnType<typeof prepareWorkflowJudgeCache>['lookup']> = null;
@@ -458,19 +465,19 @@ async function runWorkflowJudge(opts: {
evalCollector?.addTest({
name: opts.testName, suite: opts.suite, tier: 'llm-judge', passed, attempt,
duration_ms: Math.max(0, performance.now() - started),
cost_usd: reused || !scores ? 0 : 0.02,
cost_usd: reused || !samples ? 0 : 0.02 * samples.length,
execution: reused ? 'reused' : 'executed',
...customInputMetadata,
...(manualReview ? { manual_review: manualReview } : {}),
...(reused ? { reused_from: { input_key: reused.reuse.key, run_id: reused.reuse.source.runId,
revision: reused.reuse.source.revision, completed_at: new Date(reused.reuse.source.completedAt).toISOString() } } : {}),
...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning } : {}),
...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability } } : {}),
...(samples ? { judge_reasoning: judgePanelReasoning(samples) } : {}),
...(passed ? {} : { exit_reason: error instanceof JudgeRefusalError ? 'provider_refusal'
: error instanceof Error && error.name === 'WorkflowJudgeDeadline' ? 'timeout'
: error instanceof Error && error.name === 'WorkflowJudgeSuperseded' ? 'cancelled'
: stage === 'validation' ? 'validation_failed' : 'harness_error',
error: `${error instanceof Error ? error.message : String(error)}${scores ? '' : error instanceof JudgeRefusalError
error: `${error instanceof Error ? error.message : String(error)}${samples ? '' : error instanceof JudgeRefusalError
? '\nNo automated score; provider refusal usage retained when manually accepted; cost unavailable.'
: '\nNo completed model response; cost and usage unavailable.'}` }),
});
@@ -508,11 +515,11 @@ async function runWorkflowJudge(opts: {
checkActive();
stage = 'judge';
const maxTokens = opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS;
let result: JudgeScore;
let result: JudgeScore[];
try {
result = reused?.scores ?? await callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,
result = reused?.samples ?? await judgePanel(() => callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,
...(opts.stream ? { stream: true } : {}),
...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) });
...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) }));
} catch (error) {
checkActive();
if (error instanceof JudgeRefusalError && customInputMetadata) {
@@ -529,20 +536,21 @@ async function runWorkflowJudge(opts: {
throw error;
}
checkActive();
scores = result;
samples = result;
console.log(`[workflow-judge] ${opts.testName}: ${reused ? `reused ${reused.reuse.source.runId} @ ${reused.reuse.source.revision} (${new Date(reused.reuse.source.completedAt).toISOString()})` : 'executed'}`);
console.log(`${opts.testName} scores:`, JSON.stringify(scores, null, 2));
stage = 'validation';
if (opts.structuredResponse && !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true)) {
if (opts.structuredResponse && !samples.every(sample => validWorkflowJudgeScore(sample as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true))) {
throw new Error('Structured workflow judge violated the response schema');
}
scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log(`${opts.testName} panel:`, JSON.stringify({ mean: scores, samples }, null, 2));
expect(scores.clarity).toBeGreaterThanOrEqual(thresholds.clarity);
expect(scores.completeness).toBeGreaterThanOrEqual(thresholds.completeness);
expect(scores.actionability).toBeGreaterThanOrEqual(thresholds.actionability);
checkActive();
stage = 'recording';
arm();
const discardReceipt = reused ? undefined : cache.publish(scores, active);
const discardReceipt = reused ? undefined : cache.publish(samples, active);
try { checkActive(); finish(true); }
catch (error) { discardReceipt?.(); throw error; }
};
@@ -792,7 +800,7 @@ describeIfSelected('Voice directive eval', ['voice directive tone'], () => {
const voiceEnd = content.indexOf('\n## ', voiceStart + 1);
const voiceSection = content.slice(voiceStart, voiceEnd > 0 ? voiceEnd : voiceStart + 3000);
const result = await callJudge<{
const samples = await judgePanel(() => callJudge<{
directness: number;
concreteness: number;
avoids_corporate: number;
@@ -812,9 +820,10 @@ Return JSON only:
{"directness": N, "concreteness": N, "avoids_corporate": N, "avoids_ai_vocabulary": N, "connects_user_outcomes": N, "reasoning": "..."}
THE VOICE DIRECTIVE:
${voiceSection}`);
${voiceSection}`));
const result = judgePanelMean(samples, ['directness', 'concreteness', 'avoids_corporate', 'avoids_ai_vocabulary', 'connects_user_outcomes'] as const);
console.log('Voice directive scores:', JSON.stringify(result, null, 2));
console.log('Voice directive panel:', JSON.stringify({ mean: result, samples }, null, 2));
evalCollector?.addTest({
name: 'voice directive tone',
@@ -823,7 +832,7 @@ ${voiceSection}`);
passed: result.directness >= 4 && result.concreteness >= 4 && result.avoids_corporate >= 4
&& result.avoids_ai_vocabulary >= 4 && result.connects_user_outcomes >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: {
directness: result.directness,
concreteness: result.concreteness,
@@ -831,7 +840,7 @@ ${voiceSection}`);
avoids_ai_vocabulary: result.avoids_ai_vocabulary,
connects_user_outcomes: result.connects_user_outcomes,
},
judge_reasoning: result.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(result.directness).toBeGreaterThanOrEqual(4);
+2 -2
View File
@@ -292,12 +292,12 @@ test('F9 changed-input selection produces three cases with exact patterns and co
expect(prProfileTestNamePattern(files[1], selected.selection)).toBe('(?:^|\\s)(?:investigate-owned-abort|investigate-owned-ending-error)$');
});
test('both F9 files fit the existing wall with every Bun retry and reserve', () => {
test('both F9 files fit the existing wall with their one run and reserve', () => {
for (const file of files) {
const source = fs.readFileSync(path.join(import.meta.dir, '..', file), 'utf8');
const count = PR_PROFILE_FILES[file].length;
expect([...source.matchAll(/\}, CAPTURE_MS\);/g)]).toHaveLength(count);
expect(retriesForFiles([file])).toBe(1);
expect(retriesForFiles([file])).toBe(0);
const budget = resolvePaidShardBudget([file]);
expect(budget).toEqual({ timeoutMs: DEFAULT_SHARD_TIMEOUT_MS, source: 'default', policyId: null });
expect(count * CAPTURE_MS * (retriesForFiles([file]) + 1) + 120000).toBeLessThanOrEqual(budget.timeoutMs);
+146 -46
View File
@@ -1,18 +1,22 @@
import { afterEach, expect, spyOn, test } from 'bun:test';
import { afterEach, describe, expect, spyOn, test } from 'bun:test';
import { Messages } from '@anthropic-ai/sdk/resources/messages';
import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMajority, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge';
import { EVAL_POLICY } from './helpers/periodic-exclude-data';
import { getCookieWorkflowManualReview } from './helpers/cookie-workflow-manual-review';
import { resolveEvalModel } from '../lib/eval-model';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { execFileSync } from 'node:child_process';
import { prepareWorkflowJudgeCache, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
import { prepareWorkflowJudgeCache, validWorkflowJudgePanel, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA } from './helpers/workflow-judge-input';
const roots: string[] = [];
afterEach(() => { for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); });
const scores = { clarity: 4, completeness: 5, actionability: 4, reasoning: 'Concrete steps' };
const SAMPLES = JUDGE_PANEL_SAMPLES;
const panelOf = (sample: typeof scores) => Array.from({ length: SAMPLES }, () => sample);
const panel = panelOf(scores);
function fixture() {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-judge-cache-')); roots.push(root);
const files = {
@@ -53,9 +57,9 @@ function fixture() {
}
test('the audited adapter reuses only the exact completed score and original provenance', () => {
const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(scores);
const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(panel);
expect(f.entries()).toHaveLength(1);
const reused = f.cache().lookup(); expect(reused?.scores).toEqual(scores);
const reused = f.cache().lookup(); expect(reused?.samples).toEqual(panel);
expect(reused?.reuse.source.runId).toBe('free-cache-test');
expect(reused?.reuse.source.revision).toMatch(/^[a-f0-9]{40}$/);
expect(reused?.reuse.source.completedAt).toBeLessThanOrEqual(Date.now());
@@ -73,11 +77,11 @@ test('the dependency closure includes actual installed SDK bytes and local trans
});
test('release-label changes preserve reuse; other package semantics invalidate it', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
const file = path.join(f.root, 'package.json');
const original = JSON.parse(fs.readFileSync(file, 'utf8'));
fs.writeFileSync(file, JSON.stringify({ ...original, version: '2.0.0' }, null, 2));
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
for (const change of [{ scripts: { 'test:gate': 'changed command' } }, { dependencies: { 'some-sdk': '2.0.0' } }]) {
fs.writeFileSync(file, JSON.stringify({ ...original, ...change, version: '2.0.0' }));
expect(f.cache().lookup()).toBeNull();
@@ -89,7 +93,7 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in
'scripts/test-paid-shards.ts', 'scripts/test-strict-output.ts', 'scripts/eval-select.ts',
'scripts/test-pr-profile.ts', '.github/workflows/evals.yml']) {
test(`changes in ${file} require new evaluation`, () => {
const f = fixture(); f.cache().publish(scores); const target = path.join(f.root, file);
const f = fixture(); f.cache().publish(panel); const target = path.join(f.root, file);
fs.appendFileSync(target, file.endsWith('.json') ? ' ' : '\n// changed');
f.refreshPrompt(); expect(f.cache().lookup()).toBeNull();
});
@@ -98,8 +102,8 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in
test('changed sources during an attempt and mismatched actual prompt cannot publish', () => {
const f = fixture(); const before = f.cache();
fs.appendFileSync(path.join(f.root, 'example/sections/review.md'), 'new finding');
before.publish(scores); expect(f.entries()).toHaveLength(0);
f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(scores);
before.publish(panel); expect(f.entries()).toHaveLength(0);
f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(panel);
expect(f.entries()).toHaveLength(0);
});
@@ -108,52 +112,52 @@ for (const [key, value] of Object.entries({ EVALS_FRESH: '1', EVALS_TIER: 'perio
EVALS_CACHE_REPOSITORY: '', NODE_OPTIONS: '--require=unknown', BUN_OPTIONS: '--preload=unknown',
ANTHROPIC_BASE_URL: 'https://custom-provider.example.test' })) {
test(`${key}=${value} is fresh or ineligible`, () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.env = { ...f.env, [key]: value }; const cache = f.cache();
expect(cache.lookup()).toBeNull(); cache.publish(scores); expect(f.entries()).toHaveLength(1);
expect(cache.lookup()).toBeNull(); cache.publish(panel); expect(f.entries()).toHaveLength(1);
});
}
test('runtime/model/threshold changes miss, and retries never reuse or publish', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
for (const overrides of [{ GSTACK_EVAL_MODEL_JUDGE: 'different-model' }, { EVALS_CACHE_RUNTIME_ID: 'c'.repeat(64) }]) {
f.opts.env = { ...f.env, ...overrides }; expect(f.cache().lookup()).toBeNull();
}
f.opts.env = f.env; f.opts.thresholds.clarity = 5; expect(f.cache().lookup()).toBeNull();
f.opts.thresholds.clarity = 4; f.opts.attempt = 2; const retry = f.cache();
expect(retry.lookup()).toBeNull(); retry.publish(scores); expect(f.entries()).toHaveLength(1);
expect(retry.lookup()).toBeNull(); retry.publish(panel); expect(f.entries()).toHaveLength(1);
});
test('frontier reader calibration cannot reuse a score from the unspecified-reader rubric', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
const original = f.opts.prompt;
f.opts.agentCapability = 'frontier'; f.refreshPrompt();
expect(f.opts.prompt).not.toBe(original);
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(2);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
delete f.opts.agentCapability; f.refreshPrompt();
expect(f.opts.prompt).toBe(original);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('a pinned workflow judge model overrides the global model and changes the cache identity', () => {
const f = fixture();
f.opts.model = 'claude-sonnet-4-6';
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(1);
f.opts.env = { ...f.env, GSTACK_EVAL_MODEL_JUDGE: 'different-global-model' };
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
f.opts.model = 'claude-opus-4-7';
expect(f.cache().lookup()).toBeNull();
});
test('failed assertions, missing provenance, and missing imported dependencies cannot supply a receipt', () => {
const f = fixture(); f.cache().publish({ ...scores, clarity: 3 }); expect(f.entries()).toHaveLength(0);
f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(scores); expect(f.entries()).toHaveLength(0);
const f = fixture(); f.cache().publish(panelOf({ ...scores, clarity: 3 })); expect(f.entries()).toHaveLength(0);
f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(panel); expect(f.entries()).toHaveLength(0);
f.opts.env = f.env; fs.unlinkSync(path.join(f.root, 'test/helpers/nested.ts'));
f.cache().publish(scores); expect(f.entries()).toHaveLength(0);
f.cache().publish(panel); expect(f.entries()).toHaveLength(0);
});
test('cached payload schema remains small and cannot carry operational fields', () => {
@@ -167,8 +171,9 @@ test('workflow registration preserves model work and reserves only terminal-reco
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const body = source.split('async function runWorkflowJudge')[1]!.split('// Block 1:')[0]!;
const stages = ['workflowJudgeAttempts.set', 'readWorkflowJudgeInput(', 'cache.lookup()',
'callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(scores, active)']
'judgePanel(() => callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
'scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);',
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(samples, active)']
.map(stage => body.indexOf(stage));
expect(stages.every(position => position >= 0)).toBe(true);
expect(stages).toEqual([...stages].sort((a, b) => a - b));
@@ -203,6 +208,7 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
'evalCollector', 'expect', 'console', 'performance', 'JUDGE_MS', 'WORKFLOW_JUDGE_RECORD_MS',
'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'resolveEvalModel',
'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore',
'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS',
`${javascript}\nreturn runWorkflowJudge;`)(
f.root, overrides.read ?? readWorkflowJudgeInput, buildWorkflowJudgePrompt,
(options: WorkflowCacheOptions) => (overrides.prepare ?? prepareWorkflowJudgeCache)({ ...options, env: f.env }),
@@ -213,7 +219,8 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
overrides.clock ? { now: overrides.clock } : performance, overrides.budget ?? 120_000, overrides.allowance ?? 5_000,
overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout,
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore);
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore,
judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS);
return { run, records, signals, prompts, attempts, options: { ...f.opts, suite: 'Cache regression' } };
}
@@ -223,7 +230,7 @@ test('the actual workflow callback preserves the pinned model and frontier rubri
const actual = actualCallback(f, { judge: async (_prompt, model) => { models.push(model); return scores; } });
await actual.run({ ...actual.options, model: 'claude-sonnet-4-6', agentCapability: 'frontier',
readInput: () => readWorkflowJudgeInput(f.opts) });
expect(models).toEqual(['claude-sonnet-4-6']);
expect(models).toEqual(Array(SAMPLES).fill('claude-sonnet-4-6'));
expect(actual.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
expect(actual.records[0]).toMatchObject({ passed: true, model: 'claude-sonnet-4-6', prompt: actual.prompts[0] });
});
@@ -239,7 +246,7 @@ test.each(['ship', 'review'])('the registered %s callback sends the frontier rub
endMarker: f.opts.endMarker, references: [] };
const passing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 3 }) });
await passing.run(options);
expect(passing.prompts).toHaveLength(1);
expect(passing.prompts).toHaveLength(SAMPLES);
expect(passing.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
expect(passing.records[0]).toMatchObject({ passed: true, execution: 'executed', judge_scores: { clarity: 3 } });
const failing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 2 }) });
@@ -254,8 +261,8 @@ test('the actual workflow callback executes once, reuses with provenance, and pr
const f = fixture(); const first = actualCallback(f);
const options = { ...f.opts, suite: 'Cache regression' };
await first.run(options);
expect(first.prompts).toEqual([f.opts.prompt]); expect(f.entries()).toHaveLength(1);
expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 });
expect(first.prompts).toEqual(Array(SAMPLES).fill(f.opts.prompt)); expect(f.entries()).toHaveLength(1);
expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 * SAMPLES });
expect(first.records[0]).not.toHaveProperty('prompt');
expect(first.records[0]).not.toHaveProperty('model');
const reused = actualCallback(f, { judge: async () => ({ ...scores, clarity: 1 }) });
@@ -312,13 +319,13 @@ test('a superseding attempt cancels its predecessor before either can record a s
});
test('a failed input read consumes attempt one and prevents a retry from borrowing or publishing a receipt', async () => {
const f = fixture(); f.cache().publish(scores); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8');
const f = fixture(); f.cache().publish(panel); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8');
let reads = 0;
const h = actualCallback(f, { read: options => { if (++reads === 1) throw new Error('Missing workflow fixture'); return readWorkflowJudgeInput(options); } });
await expect(h.run(h.options)).rejects.toThrow('Missing workflow fixture');
expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'harness_error' });
await h.run(h.options);
expect(h.prompts).toHaveLength(1);
expect(h.prompts).toHaveLength(SAMPLES);
expect(h.records.map(record => record.execution)).toEqual(['executed', 'executed']);
expect(h.attempts.get(f.opts.testName).attempt).toBe(2);
expect(fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8')).toBe(receipt);
@@ -333,14 +340,14 @@ test('monotonic expiry after a synchronous preparation or late model response re
await expect(h.run(h.options)).rejects.toThrow('deadline');
expect(h.records).toHaveLength(1);
expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'timeout', duration_ms: 21 });
expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : 1);
expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : SAMPLES);
expect(f.entries()).toHaveLength(0);
}
});
test('publication rechecks after input scanning and withdraws a receipt if recording expires', async () => {
const f = fixture(); let checks = 0;
f.cache().publish(scores, () => ++checks < 2);
f.cache().publish(panel, () => ++checks < 2);
expect(checks).toBe(2); expect(f.entries()).toHaveLength(0);
let now = 0;
const h = actualCallback(f, { budget: 20, allowance: 5, clock: () => now,
@@ -373,7 +380,7 @@ test('the actual workflow callback preserves the complete public API body; cance
try {
const h = actualCallback(f, { judge: (prompt, model, options) => callJudge<typeof scores>(prompt, model, options) });
await h.run(h.options);
expect(create).toHaveBeenCalledTimes(1);
expect(create).toHaveBeenCalledTimes(SAMPLES);
expect(create.mock.calls[0]).toEqual([{
model: resolveEvalModel('judge'), max_tokens: 8192,
messages: [{ role: 'user', content: f.opts.prompt }],
@@ -412,40 +419,40 @@ test('Ship sends its authorized 64k cap and compact response contract through th
f.opts.structuredResponse = true;
f.opts.maxTokens = 65_536;
f.opts.stream = true;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
} finally { stream.mockRestore(); }
});
test('changing response serialization misses the cache even when prompt and model match', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.structuredResponse = true;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(2);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
const description = WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description;
try {
WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description += ' Changed response contract.';
expect(f.cache().lookup()).toBeNull();
} finally { WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description = description; }
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
f.opts.structuredResponse = false;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('the actual cap and streaming transport independently affect workflow cache identity', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.maxTokens = 65_536;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
f.opts.stream = true;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(3);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
delete f.opts.maxTokens;
delete f.opts.stream;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('the structured callback rejects incomplete, schema-invalid and below-threshold answers without cache credit', async () => {
@@ -473,3 +480,96 @@ test('the structured callback rejects incomplete, schema-invalid and below-thres
expect(validWorkflowJudgeScore({ ...scores, reasoning: Array(149).fill('word').join(' ') }, { clarity: 1, completeness: 1, actionability: 1 }, true)).toBe(true);
} finally { stream.mockRestore(); diagnostics.mockRestore(); }
});
// --- Judge panel policy (EVAL_POLICY.judge): fixed concurrent samples, per-dimension
// mean and boolean majority against unchanged thresholds, an erroring sample fails
// the whole panel and is never resampled. The provider is always a stub.
const panelScore = (clarity: number, completeness = 4, actionability = 4) => ({ clarity, completeness, actionability, reasoning: `c${clarity}` });
const panelThresholds = { clarity: 3, completeness: 3, actionability: 4 };
const panelRefusal = () => new JudgeRefusalError({ id: 'msg_1', _request_id: 'req_1', model: 'm', usage: { input_tokens: 1, output_tokens: 0 }, content: [] });
describe('judge panel', () => {
test('the pre-registered panel is three samples, and the helper restates EVAL_POLICY exactly', () => {
expect(EVAL_POLICY.judge.samples).toBe(3);
expect(JUDGE_PANEL_SAMPLES).toBe(EVAL_POLICY.judge.samples);
});
test('draws every sample concurrently before any resolves', async () => {
let started = 0;
const releases: Array<() => void> = [];
const panel = judgePanel(() => new Promise<number>(resolve => { started += 1; releases.push(() => resolve(started)); }));
await Promise.resolve();
expect(started).toBe(SAMPLES);
releases.forEach(release => release());
expect(await panel).toHaveLength(SAMPLES);
});
test('an erroring sample fails the panel and is never resampled', async () => {
let calls = 0;
const panel = judgePanel(async () => {
calls += 1;
if (calls === 2) throw new Error('Judge returned non-JSON: nope');
return panelScore(5);
});
await expect(panel).rejects.toThrow('non-JSON');
expect(calls).toBe(SAMPLES);
});
test('a refusal on every sample stays a provider refusal; a partial refusal is an ordinary failure', async () => {
await expect(judgePanel(async () => { throw panelRefusal(); })).rejects.toBeInstanceOf(JudgeRefusalError);
let calls = 0;
const partial = judgePanel(async () => { if (++calls === 1) throw panelRefusal(); return panelScore(4); });
const error = await partial.then(() => null, (reason: unknown) => reason);
expect(error).toBeInstanceOf(Error);
expect(error).not.toBeInstanceOf(JudgeRefusalError);
expect(String(error)).toContain(`sample 1 of ${SAMPLES} failed beside scored samples`);
});
test('numeric dimensions gate on the per-dimension mean; one low sample can be outvoted, a low mean cannot', () => {
const outvoted = judgePanelMean([panelScore(2), panelScore(4), panelScore(4)], JUDGE_SCORE_DIMENSIONS);
expect(outvoted.clarity).toBeCloseTo(10 / 3);
expect(outvoted.clarity).toBeGreaterThanOrEqual(panelThresholds.clarity);
const low = judgePanelMean([panelScore(2), panelScore(2), panelScore(4)], JUDGE_SCORE_DIMENSIONS);
expect(low.clarity).toBeLessThan(panelThresholds.clarity);
// No compensation across dimensions: each is averaged on its own.
expect(judgePanelMean([panelScore(5, 1), panelScore(5, 1), panelScore(5, 1)], JUDGE_SCORE_DIMENSIONS).completeness).toBe(1);
});
test('malformed sample fields fail the panel instead of averaging to NaN', () => {
expect(() => judgePanelMean([panelScore(4), { ...panelScore(4), clarity: '4' as unknown as number }, panelScore(4)], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2 has non-numeric clarity');
expect(() => judgePanelMean([panelScore(4), null as unknown as ReturnType<typeof panelScore>], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2');
expect(() => judgePanelMean([], JUDGE_SCORE_DIMENSIONS)).toThrow('no samples');
});
test('boolean fields gate on a strict majority', () => {
const vote = (...values: boolean[]) => judgePanelMajority(values.map(value => ({ ok: value })), 'ok');
expect(vote(true, true, false)).toBe(true);
expect(vote(true, false, false)).toBe(false);
expect(vote(true, false)).toBe(false);
expect(() => judgePanelMajority([{ ok: true }, { ok: 'yes' }], 'ok')).toThrow('sample 2 has non-boolean ok');
});
test('reasoning keeps every sample, numbered, even for malformed samples', () => {
expect(judgePanelReasoning([panelScore(4), null, { reasoning: 7 }])).toBe('[sample 1] c4\n[sample 2] \n[sample 3] ');
});
test('the cache stores and validates only a complete panel against the mean', () => {
expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(true);
expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(2), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), { ...panelScore(4), clarity: 6 }] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4)], prompt: 'x' }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel(panelScore(4), panelThresholds)).toBe(false);
});
test('every judge in the quality file samples through the panel, never a lone call', () => {
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const calls = [...source.matchAll(/\b(?:callJudge<[^>(]*(?:<[^>]*>[^>(]*)*>|judge)\(/g)];
expect(calls.length).toBeGreaterThanOrEqual(8);
for (const call of calls) {
expect(source.slice(Math.max(0, call.index! - 25), call.index), `unpaneled judge call at offset ${call.index}`).toMatch(/judgePanel\(\(\) => $/);
}
expect(source).not.toMatch(/\bscores\.reasoning\b|\bresult\.reasoning\b/);
});
});