feat(evals): stamp trial series identities and fit panels to the live registry

- scripts/eval-trial-series.ts stamps series_identity (eval-flake-rank's
  caseSeriesIdentities) on a report's trial-outcomes JSONL as its own step,
  keeping the history tool out of the paid runner's closure;
  TrialOutcomeRecord gains the optional series_identity field.
- Slice-count plans let a registered trial spill into an ordinary lane when
  its siblings hold every long lane, so panels never share a runner.
- Re-audited test-selection.ts (Stream B added the E2E_KINDS/BEHAVIOR_WHY
  map-diff; no new module loading) and repinned its hash.
- Detach and release floors now count trial shards (66 periodic trials in
  22 panels): periodic floor 33,821s, still under eval:bg:periodic's 67,380s.
- Coordination fixtures supply the executor's trial records.
This commit is contained in:
garrytan committed 2026-09-29 19:30:43 +00:00
1 parent d5876efa5d
commit 62fb9a255d
8 files changed
+76 -31

No files matched your search

+35
View File
@@ -0,0 +1,35 @@
#!/usr/bin/env bun
/**
* Stamp `series_identity` on a report's trial-outcomes JSONL (the pass-rates
* history key: a hash of each case's own touchfiles, GLOBAL_TOUCHFILES
* excluded; scripts/eval-flake-rank.ts caseSeriesIdentities). A separate step
* after `test-paid-shards.ts --report`, so the paid runner's closure never
* imports the history tool.
*
* Usage: bun run scripts/eval-trial-series.ts <trial-outcomes.jsonl>
*/
import * as fs from 'node:fs';
import * as path from 'node:path';
import { caseSeriesIdentities } from './eval-flake-rank';
import { formatTrialOutcomes, parseTrialOutcomes } from '../test/helpers/eval-store';
const ROOT = path.resolve(import.meta.dir, '..');
/** Rewrite the file with every record stamped; an invalid line fails the whole stamp. */
export function stampTrialSeries(file: string, root = ROOT): number {
const { records, errors } = parseTrialOutcomes(fs.readFileSync(file, 'utf8'));
if (errors.length) throw new Error(`${file}: ${errors.join('; ')}`);
const identities = caseSeriesIdentities([...new Set(records.map(record => record.case))], root);
const stamped = records.map(record => ({ ...record, series_identity: identities[record.case] }));
fs.writeFileSync(file, formatTrialOutcomes(stamped));
return stamped.length;
}
if (import.meta.main) {
const file = process.argv[2];
if (!file) {
console.error('usage: bun run scripts/eval-trial-series.ts <trial-outcomes.jsonl>');
process.exit(2);
}
console.log(`[eval-trial-series] stamped ${stampTrialSeries(file)} record(s) in ${file}`);
}
+14 -20
View File
@@ -1739,11 +1739,18 @@ export function buildRunManifest(opts: {
ordinary.filter(files => !registeredFiles.has(files[0])))) {
const lanes = registeredFiles.has(files[0]) ? longLanes : ordinarySlices;
const laneKeys = (index: number) => [...allocations].filter(([, lane]) => lane === index + 1).map(([key]) => key);
let lane = -1;
for (let index = 0; index < lanes; index++) {
if (sharesPanel(laneKeys(index), files[0]!)) continue;
if (lane < 0 || loads[index] < loads[lane]) lane = index;
}
// A trial whose siblings already hold every long lane may use any
// ordinary lane: independent runners outrank long-lane ownership.
const pick = (limit: number) => {
let best = -1;
for (let index = 0; index < limit; index++) {
if (sharesPanel(laneKeys(index), files[0]!)) continue;
if (best < 0 || loads[index] < loads[best]) best = index;
}
return best;
};
let lane = pick(lanes);
if (lane < 0) lane = pick(ordinarySlices);
if (lane < 0) lane = loads.slice(0, lanes).indexOf(Math.min(...loads.slice(0, lanes)));
allocations.set(files[0], lane + 1);
loads[lane] += resolvePaidShardTimeoutMs(files, opts.timeoutMs);
@@ -2500,21 +2507,7 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo
const laterPanels = attempts.slice(1).flatMap(attempt => panelReports(manifest, artifacts.map(a => a.result), attempt)
.filter(panel => panel.trials.length > 0));
// Quarantine policy checks on census runs: the per-tier cap and entry expiry.
if (manifest.evalsAll) {
const tierIds = Object.keys(E2E_TIERS).filter(id => E2E_TIERS[id] === manifest.tier);
const quarantined = Object.keys(CASE_QUARANTINE).filter(id => E2E_TIERS[id] === manifest.tier);
if (quarantined.length > EVAL_POLICY.quarantine.capFraction * tierIds.length) {
verdict.problems.push(`QUARANTINE over cap: ${quarantined.length} of ${tierIds.length} ${manifest.tier} cases (cap ${Math.round(EVAL_POLICY.quarantine.capFraction * 100)}%)`);
}
const expiryMs = EVAL_POLICY.quarantine.expiryWeeklyRuns * 7 * 24 * 60 * 60 * 1000;
for (const id of quarantined) {
const entered = Date.parse(CASE_QUARANTINE[id]!.enteredAt);
if (!Number.isFinite(entered) || Date.now() - entered > expiryMs) {
verdict.problems.push(`QUARANTINE expired: ${id} (entered ${CASE_QUARANTINE[id]!.enteredAt}; entries expire after ${EVAL_POLICY.quarantine.expiryWeeklyRuns} weekly runs)`);
}
}
}
// Quarantine cap and expiry are the weekly pass-rates gate's (eval-flake-rank --gate).
// History: one trial-outcomes line per isolated trial and per JUnit rule/judge case.
const runId = env.GITHUB_RUN_ID;
@@ -2567,6 +2560,7 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo
}
}
}
// series_identity is stamped afterwards by scripts/eval-trial-series.ts (the report job's next step).
fs.writeFileSync(trialOutcomesPath, formatTrialOutcomes(history));
// Headline and failure block (A4): one formatter for the log, the PR comment and the weekly issue.
+8 -2
View File
@@ -3,7 +3,7 @@ import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { buildRunManifest, collectPaidTestFiles, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards';
import { buildRunManifest, collectPaidTestFiles, shardCaseId, shardTrial, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards';
import { STRICT_RETRY_CASE_BUDGETS } from './helpers/eval-budgets';
import { approvedCookieWorkflowSource, manualReviewFixture } from './helpers/manual-judge-review-fixture';
@@ -16,6 +16,11 @@ type Job = {
permissions: Record<string, string>;
steps: Step[];
};
/** A passing trial record for an isolated trial shard (the executor's current result schema). */
const trialRecord = (entry: PaidRunManifest['entries'][number]) => entry.trial ? { trial: {
case: shardCaseId(entry.file)!, trial: shardTrial(entry.file)!, ...entry.trial, outcome: 'passed' as const, cost_usd: 0, duration_ms: 1,
} } : {};
const workflows = ['evals.yml', 'evals-periodic.yml'].map(name => ({
name,
jobs: (Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows', name), 'utf8')) as {
@@ -217,6 +222,7 @@ describe('dependency-free CI planner and report execution', () => {
executedTests: STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === entry.file)?.cases ?? 1,
skippedTests: 0,
...(entry.budget ? { budget: entry.budget } : {}),
...trialRecord(entry),
})),
};
fs.writeFileSync(path.join(reportDir, `slice-${sliceIndex}.json`), JSON.stringify(result));
@@ -275,7 +281,7 @@ describe('dependency-free CI planner and report execution', () => {
outcomes: manifest.entries.filter(entry => entry.status === 'planned').map(entry => ({
files: [entry.file], status: 'passed', exitCode: 0, elapsedMs: 1,
executedTests: STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === entry.file)?.cases ?? 1,
skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}),
skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}), ...trialRecord(entry),
})),
};
const slicePath = path.join(reportDir, 'slice-1.json');
+6 -4
View File
@@ -1,5 +1,5 @@
import { expect, test } from 'bun:test';
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards';
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, expandTrialShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards';
import { FINDING_RETRY_BUDGETS, ALL_TIERS, SHARD_RESERVE_MS } from './helpers/eval-budgets';
import fs from 'node:fs';
import os from 'node:os';
@@ -175,8 +175,9 @@ test('single-slice manifest retains all registered files with one allocation', (
test('current detach supervision covers the live-census floor', () => {
const floorFor = (tier: 'gate' | 'periodic') => {
// Case-sharded files contribute one shard per case, exactly as the runner plans.
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier);
// Case-sharded files contribute one shard per case and isolated cases one
// shard per trial, exactly as the runner plans.
const files = expandTrialShards(expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier), tier).keys;
const excess = files.reduce((n, file) => n + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - DEFAULT_SHARD_TIMEOUT_MS), 0);
return Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05);
};
@@ -186,7 +187,8 @@ test('current detach supervision covers the live-census floor', () => {
expect(floorFor('gate')).toBe(21_725);
expect(gateTimeout).toBe(49_320);
expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate'));
expect(floorFor('periodic')).toBe(22_481);
expect(floorFor('periodic')).toBe(33_821);
expect(periodicTimeout).toBeGreaterThanOrEqual(floorFor('periodic'));
});
for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => {
+3
View File
@@ -402,6 +402,8 @@ export interface TrialOutcomeRecord {
sha?: string;
lane?: string;
recorded_at?: string;
/** History series key: a hash of the case's own touchfiles (GLOBAL_TOUCHFILES excluded), stamped by the report job. */
series_identity?: string;
}
/** First line of free text, stripped of @-mentions and control characters, capped. */
@@ -434,6 +436,7 @@ function trialRecordProblems(r: any): string[] {
if (r.execution !== 'executed' && r.execution !== 'reused') problems.push('execution invalid');
if (!['shard', 'junit', 'backfill'].includes(r.source)) problems.push('source invalid');
if (r.error !== undefined && (typeof r.error !== 'string' || r.error.length > TRIAL_ERROR_MAX)) problems.push('error invalid');
if (r.series_identity !== undefined && (typeof r.series_identity !== 'string' || !/^[\w.-]{1,64}$/.test(r.series_identity))) problems.push('series_identity invalid');
return problems;
}
+1 -1
View File
@@ -27,7 +27,7 @@ function runnerDependencies(root: string, entries: string[]): string[] {
const source = fs.readFileSync(file, 'utf8').replace(/^#![^\n]*(?:\n|$)/, '\n');
let audited = source;
if (relative === 'test/helpers/test-selection.ts') {
if (createHash('sha256').update(source).digest('hex') !== '4d2fbcb6249e8d22453d25bfe9b18ee0f4568bbec071918675c38a455d4e1e08') {
if (createHash('sha256').update(source).digest('hex') !== '052ad5a52472bcb41db04c9f21fe6a819e9547768468f7e5d390e0014b567677') {
throw new Error('Re-audit the historical touchfile map loader before excluding its computed import');
}
audited = source.replace('`const m = await import(${JSON.stringify(dataPath)});`,', "'',");
+7 -2
View File
@@ -11,6 +11,7 @@ import * as os from 'node:os';
import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { parseRunManifest, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards';
import { stampTrialSeries } from '../scripts/eval-trial-series';
const ROOT = path.resolve(import.meta.dir, '..');
const RULE_A = 'test/skill-e2e-fail-open-alpha.test.ts';
@@ -148,8 +149,12 @@ describe('behavior and quarantined panels through --report', () => {
const summary = JSON.parse(fs.readFileSync(path.join(r.dir, 'collector-outcomes.json'), 'utf8'));
expect(summary.version).toBe(2);
expect(summary.panels[0]).toMatchObject({ case: ID, status: 'PASS', split: true, failsLane: false });
const history = fs.readFileSync(path.join(r.dir, 'trial-outcomes.jsonl'), 'utf8').trim().split('\n').map(line => JSON.parse(line));
expect(history.map(h => [h.trial, h.outcome])).toEqual([[1, 'passed'], [2, 'failed'], [3, 'passed']]);
const outcomesFile = path.join(r.dir, 'trial-outcomes.jsonl');
const history = () => fs.readFileSync(outcomesFile, 'utf8').trim().split('\n').map(line => JSON.parse(line));
expect(history().map(h => [h.trial, h.outcome])).toEqual([[1, 'passed'], [2, 'failed'], [3, 'passed']]);
expect(stampTrialSeries(outcomesFile)).toBe(3);
expect(new Set(history().map(h => h.series_identity)).size).toBe(1);
expect(history()[0].series_identity).toMatch(/^[0-9a-f]{16}$/);
});
test('behavior 1/3: red', () => {
+2 -2
View File
@@ -230,8 +230,8 @@ test('detached PR fallback and release commands cover their actual default worke
)) / 1000 * 1.05));
}
const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000;
expect(releaseFloors).toEqual([21_725, 22_481]);
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(44_206);
expect(releaseFloors).toEqual([21_725, 33_821]);
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(55_546);
expect(detachedReleaseWall).toBe(116_700_000);
expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000);
});