feat(evals): stamp trial series identities and fit panels to the live registry

- scripts/eval-trial-series.ts stamps series_identity (eval-flake-rank's
  caseSeriesIdentities) on a report's trial-outcomes JSONL as its own step,
  keeping the history tool out of the paid runner's closure;
  TrialOutcomeRecord gains the optional series_identity field.
- Slice-count plans let a registered trial spill into an ordinary lane when
  its siblings hold every long lane, so panels never share a runner.
- Re-audited test-selection.ts (Stream B added the E2E_KINDS/BEHAVIOR_WHY
  map-diff; no new module loading) and repinned its hash.
- Detach and release floors now count trial shards (66 periodic trials in
  22 panels): periodic floor 33,821s, still under eval:bg:periodic's 67,380s.
- Coordination fixtures supply the executor's trial records.
This commit is contained in:
garrytan committed 2026-09-29 19:30:43 +00:00
1 parent d5876efa5d
commit 62fb9a255d
8 files changed
+76 -31

No files matched your search

+35
View File
@@ -0,0 +1,35 @@
#!/usr/bin/env bun
/**
* Stamp `series_identity` on a report's trial-outcomes JSONL (the pass-rates
* history key: a hash of each case's own touchfiles, GLOBAL_TOUCHFILES
* excluded; scripts/eval-flake-rank.ts caseSeriesIdentities). A separate step
* after `test-paid-shards.ts --report`, so the paid runner's closure never
* imports the history tool.
*
* Usage: bun run scripts/eval-trial-series.ts <trial-outcomes.jsonl>
*/
import * as fs from 'node:fs';
import * as path from 'node:path';
import { caseSeriesIdentities } from './eval-flake-rank';
import { formatTrialOutcomes, parseTrialOutcomes } from '../test/helpers/eval-store';
const ROOT = path.resolve(import.meta.dir, '..');
/** Rewrite the file with every record stamped; an invalid line fails the whole stamp. */
export function stampTrialSeries(file: string, root = ROOT): number {
const { records, errors } = parseTrialOutcomes(fs.readFileSync(file, 'utf8'));
if (errors.length) throw new Error(`${file}: ${errors.join('; ')}`);
const identities = caseSeriesIdentities([...new Set(records.map(record => record.case))], root);
const stamped = records.map(record => ({ ...record, series_identity: identities[record.case] }));
fs.writeFileSync(file, formatTrialOutcomes(stamped));
return stamped.length;
}
if (import.meta.main) {
const file = process.argv[2];
if (!file) {
console.error('usage: bun run scripts/eval-trial-series.ts <trial-outcomes.jsonl>');
process.exit(2);
}
console.log(`[eval-trial-series] stamped ${stampTrialSeries(file)} record(s) in ${file}`);
}
+14 -20
View File
@@ -1739,11 +1739,18 @@ export function buildRunManifest(opts: {
ordinary.filter(files => !registeredFiles.has(files[0])))) {
const lanes = registeredFiles.has(files[0]) ? longLanes : ordinarySlices;
const laneKeys = (index: number) => [...allocations].filter(([, lane]) => lane === index + 1).map(([key]) => key);
let lane = -1;
for (let index = 0; index < lanes; index++) {
if (sharesPanel(laneKeys(index), files[0]!)) continue;
if (lane < 0 || loads[index] < loads[lane]) lane = index;
}
// A trial whose siblings already hold every long lane may use any
// ordinary lane: independent runners outrank long-lane ownership.
const pick = (limit: number) => {
let best = -1;
for (let index = 0; index < limit; index++) {
if (sharesPanel(laneKeys(index), files[0]!)) continue;
if (best < 0 || loads[index] < loads[best]) best = index;
}
return best;
};
let lane = pick(lanes);
if (lane < 0) lane = pick(ordinarySlices);
if (lane < 0) lane = loads.slice(0, lanes).indexOf(Math.min(...loads.slice(0, lanes)));
allocations.set(files[0], lane + 1);
loads[lane] += resolvePaidShardTimeoutMs(files, opts.timeoutMs);
@@ -2500,21 +2507,7 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo
const laterPanels = attempts.slice(1).flatMap(attempt => panelReports(manifest, artifacts.map(a => a.result), attempt)
.filter(panel => panel.trials.length > 0));
// Quarantine policy checks on census runs: the per-tier cap and entry expiry.
if (manifest.evalsAll) {
const tierIds = Object.keys(E2E_TIERS).filter(id => E2E_TIERS[id] === manifest.tier);
const quarantined = Object.keys(CASE_QUARANTINE).filter(id => E2E_TIERS[id] === manifest.tier);
if (quarantined.length > EVAL_POLICY.quarantine.capFraction * tierIds.length) {
verdict.problems.push(`QUARANTINE over cap: ${quarantined.length} of ${tierIds.length} ${manifest.tier} cases (cap ${Math.round(EVAL_POLICY.quarantine.capFraction * 100)}%)`);
}
const expiryMs = EVAL_POLICY.quarantine.expiryWeeklyRuns * 7 * 24 * 60 * 60 * 1000;
for (const id of quarantined) {
const entered = Date.parse(CASE_QUARANTINE[id]!.enteredAt);
if (!Number.isFinite(entered) || Date.now() - entered > expiryMs) {
verdict.problems.push(`QUARANTINE expired: ${id} (entered ${CASE_QUARANTINE[id]!.enteredAt}; entries expire after ${EVAL_POLICY.quarantine.expiryWeeklyRuns} weekly runs)`);
}
}
}
// Quarantine cap and expiry are the weekly pass-rates gate's (eval-flake-rank --gate).
// History: one trial-outcomes line per isolated trial and per JUnit rule/judge case.
const runId = env.GITHUB_RUN_ID;
@@ -2567,6 +2560,7 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo
}
}
}
// series_identity is stamped afterwards by scripts/eval-trial-series.ts (the report job's next step).
fs.writeFileSync(trialOutcomesPath, formatTrialOutcomes(history));
// Headline and failure block (A4): one formatter for the log, the PR comment and the weekly issue.