mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
feat(evals): planner-side whole-panel reuse and negative receipts
The planner job restores this PR's receipt store once and ships a single filtered set with the plan: a pass or panel receipt with a same-or-newer FAIL for its input identity is dropped, and a panel receipt ships only as a whole PASS panel (re-verified with panelVerdict) from one run. Executors read only that set (no per-slice cache restore or save), so every trial of a panel sees the same receipts; a trial reuses its own record from the panel receipt, keeping a split PASS's failed trial. Trial identities drop the trial index (run-scoped) and bind the panel policy. Executed shards carry their input identity; the report turns a whole fresh PASS panel into a panel receipt and a FAIL panel or failed rule shard into a negative receipt, and marks a panel that mixes reused and fresh trials INCOMPLETE. The report job merges plan, slice and report receipts (newest per file) and saves one store per run. Also fixes two TS2352 casts in browse/test/dia-macos-qualification.test.ts whose diagnostic text drifted with program order (baseline locked, fix only).
This commit is contained in:
1 parent
8622535b90
commit
3bae8e33da
9 files changed
+410
-110
No files matched your search
+42
-56
@@ -21,19 +21,31 @@ test('only PR runs select the fast profile; manual and scheduled coverage stays
|
||||
}
|
||||
});
|
||||
|
||||
test('receipt transport restores only this repository and PR with no broad fallback key', () => {
|
||||
const steps = paid.jobs['eval-slices'].steps;
|
||||
const restore = steps.filter((s: any) => s.uses?.startsWith('actions/cache/restore@'));
|
||||
const save = steps.filter((s: any) => s.uses?.startsWith('actions/cache/save@'));
|
||||
test('receipt transport: the planner restores only this repository and PR, the report saves one merged store', () => {
|
||||
const planner = paid.jobs['plan-slices'].steps;
|
||||
const restore = planner.filter((s: any) => s.uses?.startsWith('actions/cache/restore@'));
|
||||
expect(restore).toHaveLength(1);
|
||||
expect(save).toHaveLength(1);
|
||||
expect(restore[0].if).toBe("github.event_name == 'pull_request'");
|
||||
expect(restore[0].with.path).toBe('/tmp/gstack-eval-input-cache');
|
||||
expect(restore[0].with['restore-keys']).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-');
|
||||
expect(save[0].with.key).toBe(restore[0].with.key);
|
||||
expect(save[0].with.key).toContain('${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}');
|
||||
const emit = planner.find((s: any) => s.run?.includes('--emit-plan /tmp/paid-plan/manifest.json'));
|
||||
expect(emit.env.EVALS_CACHE_DIR).toBe("${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }}");
|
||||
const upload = planner.find((s: any) => s.with?.name === 'paid-plan');
|
||||
expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/paid-plan/manifest.json', '/tmp/paid-plan/receipts']);
|
||||
// Executors never restore or save a cache of their own: every slice sees the plan's one receipt set.
|
||||
const executor = paid.jobs['eval-slices'].steps;
|
||||
expect(executor.filter((s: any) => s.uses?.startsWith('actions/cache/'))).toHaveLength(0);
|
||||
expect(executor.find((s: any) => s.name === "Seed this slice's receipts from the plan").run).toContain('cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/');
|
||||
const report = paid.jobs['slices-report'].steps;
|
||||
const merge = report.find((s: any) => s.name === "Merge this run's receipts");
|
||||
expect(merge.run).toContain('scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache');
|
||||
const save = report.filter((s: any) => s.uses?.startsWith('actions/cache/save@'));
|
||||
expect(save).toHaveLength(1);
|
||||
expect(save[0].with.path).toBe('/tmp/gstack-eval-input-cache');
|
||||
expect(save[0].if).toContain("steps.receipts.outputs.present == 'true'");
|
||||
expect(save[0].with.key).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged');
|
||||
expect(report.indexOf(save[0])).toBeGreaterThan(report.indexOf(merge));
|
||||
expect(paid.jobs['eval-slices'].permissions).toEqual({ contents: 'read', packages: 'read' });
|
||||
expect(paid.jobs['slices-report'].permissions).toEqual({ contents: 'read' });
|
||||
expect(JSON.stringify(periodic)).not.toContain('actions/cache/');
|
||||
});
|
||||
|
||||
@@ -43,56 +55,21 @@ test('the judge binds cache receipts to the PR and installed runtime, not the co
|
||||
expect(runtime.run).toContain('sha256sum /tmp/eval-runtime-manifest.json');
|
||||
const run = paid.jobs['eval-slices'].steps.find((s: any) => s.run?.includes('--plan /tmp/paid-plan/manifest.json'));
|
||||
expect(run.env).toMatchObject({
|
||||
EVALS_CACHE_DIR: '/tmp/gstack-eval-input-cache',
|
||||
EVALS_CACHE_DIR: '/tmp/paid-slice-results/receipts',
|
||||
EVALS_CACHE_REPOSITORY: '${{ github.repository }}',
|
||||
EVALS_CACHE_PR: '${{ github.event.pull_request.number }}',
|
||||
EVALS_CACHE_RUNTIME_ID: '${{ needs.build-image.outputs.runtime-id }}',
|
||||
});
|
||||
});
|
||||
|
||||
test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('only a new passing producer can publish the next cache snapshot', () => {
|
||||
const directory = mkdtempSync(join(tmpdir(), 'ci-cache-producer-'));
|
||||
const receipts = join(directory, 'receipts');
|
||||
const output = join(directory, 'output');
|
||||
mkdirSync(receipts);
|
||||
const step = paid.jobs['eval-slices'].steps.find((s: any) => s.id === 'receipts');
|
||||
const script = step.run.replaceAll('/tmp/gstack-eval-input-cache', receipts);
|
||||
const run = () => {
|
||||
writeFileSync(output, '');
|
||||
const result = spawnSync('bash', ['-e', '-c', script], {
|
||||
env: { ...process.env, GITHUB_OUTPUT: output, GITHUB_RUN_ID: '42', GITHUB_RUN_ATTEMPT: '2' },
|
||||
encoding: 'utf8', timeout: 5000,
|
||||
});
|
||||
expect(result.status, result.stderr).toBe(0);
|
||||
return readFileSync(output, 'utf8');
|
||||
};
|
||||
try {
|
||||
expect(run()).toBe('');
|
||||
writeFileSync(join(receipts, 'old.json'), JSON.stringify({ proof: { source: { runId: '41/1' } } }));
|
||||
writeFileSync(join(receipts, 'corrupt.json'), '{');
|
||||
expect(run()).toBe('');
|
||||
writeFileSync(join(receipts, 'prior-attempt.json'), JSON.stringify({ proof: { source: { runId: '42/1' } } }));
|
||||
expect(run()).toBe('');
|
||||
writeFileSync(join(receipts, 'fresh.json'), JSON.stringify({ proof: { source: { runId: '42/2' } } }));
|
||||
expect(run()).toBe('present=true\n');
|
||||
} finally { rmSync(directory, { recursive: true, force: true }); }
|
||||
});
|
||||
|
||||
test.skipIf(!Bun.which('jq'))('the actual comment separates reused evidence, retry outcomes and deferred coverage', () => {
|
||||
test.skipIf(!Bun.which('jq'))('the actual comment shows deferred coverage and never recomputes a verdict', () => {
|
||||
const comment = paid.jobs['slices-comment'].steps.find((s: any) => s.name === 'Post PR comment').run as string;
|
||||
const evaluate = (filter: string, value: unknown) => {
|
||||
const result = spawnSync('jq', ['-r', filter], { input: JSON.stringify(value), encoding: 'utf8', timeout: 5000 });
|
||||
expect(result.status, result.stderr).toBe(0);
|
||||
return result.stdout.trim();
|
||||
};
|
||||
const stats = comment.match(/STATS=\$\(jq -r '([^']+)'/)![1]!;
|
||||
expect(evaluate(stats, { tests: [
|
||||
{ name: 'retry', passed: false }, { name: 'retry', passed: true },
|
||||
{ name: 'exhausted', passed: false }, { name: 'exhausted', passed: false },
|
||||
{ name: 'regressed', passed: true }, { name: 'regressed', passed: false },
|
||||
{ name: 'reused', passed: true, execution: 'reused' },
|
||||
], flaky_retries: ['retry', 'exhausted', 'regressed'].map(name => ({ name, attempts: 2 })) })).toBe('4 2 2 3 3 1');
|
||||
expect(comment).toContain("printf ' | ⚠ %s cases with multiple attempts'");
|
||||
expect(comment).not.toContain('group_by(.name)');
|
||||
expect(comment).not.toMatch(/flaky pass\(es\)|passed only on retry|not blocking/);
|
||||
const coverage = comment.match(/COVERAGE=\$\(jq -r '([^']+)'/)![1]!;
|
||||
const text = evaluate(coverage, { profile: 'pr', selection: { e2e: ['probe'], judges: ['judge'] },
|
||||
@@ -107,9 +84,9 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
|
||||
const job = paid.jobs['slices-comment'];
|
||||
expect(job.permissions).toMatchObject({ 'pull-requests': 'write' });
|
||||
expect(JSON.stringify(job.steps)).not.toMatch(/actions\/checkout|setup-bun|bun run|npm |node /);
|
||||
const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict');
|
||||
expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json']);
|
||||
expect(job.steps.find((step: any) => step.with?.name === 'report-verdict').with.path).toBe('/tmp/verdict');
|
||||
const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}');
|
||||
expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json', '/tmp/paid-report/report-summary.md']);
|
||||
expect(job.steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}').with.path).toBe('/tmp/verdict');
|
||||
const root = mkdtempSync(join(tmpdir(), 'ci-comment-'));
|
||||
const paidDir = join(root, 'paid-report');
|
||||
const verdictDir = join(root, 'verdict');
|
||||
@@ -123,9 +100,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
|
||||
writeFileSync(join(paidDir, 'judge.json'), JSON.stringify({ total_tests: 2, tier: 'llm-judge', shard: 1,
|
||||
tests: [{ name: 'manual', passed: false, manual_review: { unverified: true } },
|
||||
{ name: 'reused', passed: true, execution: 'reused' }], flaky_retries: [] }));
|
||||
const summary = { version: 1, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0,
|
||||
const summary = { version: 2, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0,
|
||||
total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 }],
|
||||
totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 } };
|
||||
totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 },
|
||||
verdict: { verdict: 'GREEN' }, headline: ['[test:paid] VERDICT GREEN — lane gate/pr, attempt 1'], panels: [],
|
||||
failures: ['⚠ case-x behavior PASS 2/3 (✓✗✓) t2: timeout at turn 3 — @\u200bsomeone said no'] };
|
||||
mkdirSync(join(verdictDir, 'paid-report'));
|
||||
const summaryPath = join(verdictDir, 'paid-report/collector-outcomes.json');
|
||||
const script = (job.steps.find((step: any) => step.name === 'Post PR comment').run as string)
|
||||
@@ -142,8 +121,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
|
||||
const verified = run();
|
||||
expect(verified.status, verified.stderr).toBe(0);
|
||||
expect(verified.stdout).toContain('⚠ MANUAL ACCEPTED (unscored)');
|
||||
expect(verified.stdout).toContain('1 automated passed / 2 final results');
|
||||
expect(verified.stdout).toContain('0 failed, 1 manual accepted');
|
||||
expect(verified.stdout).toContain('VERDICT GREEN — lane gate/pr, attempt 1');
|
||||
expect(verified.stdout).toContain('1 executed, 1 reused** rule/judge records');
|
||||
expect(verified.stdout).toContain('1 manual accepted');
|
||||
expect(verified.stdout).toContain('### Failures and split verdicts');
|
||||
expect(verified.stdout).toContain('PASS 2/3 (✓✗✓) t2: timeout at turn 3');
|
||||
|
||||
const unrelatedFailure = { ...summary, files: [{ ...summary.files[0], total: 3, failed: 1,
|
||||
executed: 2, attempts: 3 }], totals: { ...summary.totals, total: 3, failed: 1,
|
||||
@@ -152,16 +134,20 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
|
||||
const red = run();
|
||||
expect(red.status, red.stderr).toBe(0);
|
||||
expect(red.stdout).toContain('❌ FAIL');
|
||||
expect(red.stdout).toContain('1 failed, 1 manual accepted');
|
||||
|
||||
writeFileSync(summaryPath, JSON.stringify({ ...summary, verdict: { verdict: 'RED' } }));
|
||||
const redVerdict = run();
|
||||
expect(redVerdict.status, redVerdict.stderr).toBe(0);
|
||||
expect(redVerdict.stdout).toContain('❌ FAIL');
|
||||
|
||||
writeFileSync(summaryPath, JSON.stringify({ ...summary, totals: { ...summary.totals, manual_accepted: 2 } }));
|
||||
const tampered = run();
|
||||
expect(tampered.status, tampered.stderr).toBe(0);
|
||||
expect(tampered.stdout).toContain('manual acceptance unavailable/unverified');
|
||||
expect(tampered.stdout).toContain('verified report unavailable');
|
||||
expect(tampered.stdout).not.toContain('⚠ MANUAL ACCEPTED (unscored)');
|
||||
rmSync(summaryPath);
|
||||
const absent = run();
|
||||
expect(absent.status, absent.stderr).toBe(0);
|
||||
expect(absent.stdout).toContain('manual acceptance unavailable/unverified');
|
||||
expect(absent.stdout).toContain('verified report unavailable');
|
||||
} finally { rmSync(root, { recursive: true, force: true }); }
|
||||
});
|
||||
Reference in new issue
Block a user