feat(evals): stamp trial series identities and fit panels to the live registry

- scripts/eval-trial-series.ts stamps series_identity (eval-flake-rank's
  caseSeriesIdentities) on a report's trial-outcomes JSONL as its own step,
  keeping the history tool out of the paid runner's closure;
  TrialOutcomeRecord gains the optional series_identity field.
- Slice-count plans let a registered trial spill into an ordinary lane when
  its siblings hold every long lane, so panels never share a runner.
- Re-audited test-selection.ts (Stream B added the E2E_KINDS/BEHAVIOR_WHY
  map-diff; no new module loading) and repinned its hash.
- Detach and release floors now count trial shards (66 periodic trials in
  22 panels): periodic floor 33,821s, still under eval:bg:periodic's 67,380s.
- Coordination fixtures supply the executor's trial records.
This commit is contained in:
garrytan committed 2026-09-29 19:30:43 +00:00
1 parent d5876efa5d
commit 62fb9a255d
8 files changed
+76 -31

No files matched your search

+14 -20
View File
@@ -1739,11 +1739,18 @@ export function buildRunManifest(opts: {
ordinary.filter(files => !registeredFiles.has(files[0])))) {
const lanes = registeredFiles.has(files[0]) ? longLanes : ordinarySlices;
const laneKeys = (index: number) => [...allocations].filter(([, lane]) => lane === index + 1).map(([key]) => key);
let lane = -1;
for (let index = 0; index < lanes; index++) {
if (sharesPanel(laneKeys(index), files[0]!)) continue;
if (lane < 0 || loads[index] < loads[lane]) lane = index;
}
// A trial whose siblings already hold every long lane may use any
// ordinary lane: independent runners outrank long-lane ownership.
const pick = (limit: number) => {
let best = -1;
for (let index = 0; index < limit; index++) {
if (sharesPanel(laneKeys(index), files[0]!)) continue;
if (best < 0 || loads[index] < loads[best]) best = index;
}
return best;
};
let lane = pick(lanes);
if (lane < 0) lane = pick(ordinarySlices);
if (lane < 0) lane = loads.slice(0, lanes).indexOf(Math.min(...loads.slice(0, lanes)));
allocations.set(files[0], lane + 1);
loads[lane] += resolvePaidShardTimeoutMs(files, opts.timeoutMs);
@@ -2500,21 +2507,7 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo
const laterPanels = attempts.slice(1).flatMap(attempt => panelReports(manifest, artifacts.map(a => a.result), attempt)
.filter(panel => panel.trials.length > 0));
// Quarantine policy checks on census runs: the per-tier cap and entry expiry.
if (manifest.evalsAll) {
const tierIds = Object.keys(E2E_TIERS).filter(id => E2E_TIERS[id] === manifest.tier);
const quarantined = Object.keys(CASE_QUARANTINE).filter(id => E2E_TIERS[id] === manifest.tier);
if (quarantined.length > EVAL_POLICY.quarantine.capFraction * tierIds.length) {
verdict.problems.push(`QUARANTINE over cap: ${quarantined.length} of ${tierIds.length} ${manifest.tier} cases (cap ${Math.round(EVAL_POLICY.quarantine.capFraction * 100)}%)`);
}
const expiryMs = EVAL_POLICY.quarantine.expiryWeeklyRuns * 7 * 24 * 60 * 60 * 1000;
for (const id of quarantined) {
const entered = Date.parse(CASE_QUARANTINE[id]!.enteredAt);
if (!Number.isFinite(entered) || Date.now() - entered > expiryMs) {
verdict.problems.push(`QUARANTINE expired: ${id} (entered ${CASE_QUARANTINE[id]!.enteredAt}; entries expire after ${EVAL_POLICY.quarantine.expiryWeeklyRuns} weekly runs)`);
}
}
}
// Quarantine cap and expiry are the weekly pass-rates gate's (eval-flake-rank --gate).
// History: one trial-outcomes line per isolated trial and per JUnit rule/judge case.
const runId = env.GITHUB_RUN_ID;
@@ -2567,6 +2560,7 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo
}
}
}
// series_identity is stamped afterwards by scripts/eval-trial-series.ts (the report job's next step).
fs.writeFileSync(trialOutcomesPath, formatTrialOutcomes(history));
// Headline and failure block (A4): one formatter for the log, the PR comment and the weekly issue.