From 62fb9a255d3c9737cbe390d6b5e2ab0e4a4fffd3 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:30:43 +0000 Subject: [PATCH] feat(evals): stamp trial series identities and fit panels to the live registry - scripts/eval-trial-series.ts stamps series_identity (eval-flake-rank's caseSeriesIdentities) on a report's trial-outcomes JSONL as its own step, keeping the history tool out of the paid runner's closure; TrialOutcomeRecord gains the optional series_identity field. - Slice-count plans let a registered trial spill into an ordinary lane when its siblings hold every long lane, so panels never share a runner. - Re-audited test-selection.ts (Stream B added the E2E_KINDS/BEHAVIOR_WHY map-diff; no new module loading) and repinned its hash. - Detach and release floors now count trial shards (66 periodic trials in 22 panels): periodic floor 33,821s, still under eval:bg:periodic's 67,380s. - Coordination fixtures supply the executor's trial records. --- scripts/eval-trial-series.ts | 35 +++++++++++++++++++++++++++ scripts/test-paid-shards.ts | 34 +++++++++++--------------- test/ci-paid-coordination.test.ts | 10 ++++++-- test/eng-finding-retry-budget.test.ts | 10 +++++--- test/helpers/eval-store.ts | 3 +++ test/paid-free-boundary.test.ts | 2 +- test/paid-report-fail-open.test.ts | 9 +++++-- test/paid-retry-supervision.test.ts | 4 +-- 8 files changed, 76 insertions(+), 31 deletions(-) create mode 100644 scripts/eval-trial-series.ts diff --git a/scripts/eval-trial-series.ts b/scripts/eval-trial-series.ts new file mode 100644 index 000000000..e6b370f83 --- /dev/null +++ b/scripts/eval-trial-series.ts @@ -0,0 +1,35 @@ +#!/usr/bin/env bun +/** + * Stamp `series_identity` on a report's trial-outcomes JSONL (the pass-rates + * history key: a hash of each case's own touchfiles, GLOBAL_TOUCHFILES + * excluded; scripts/eval-flake-rank.ts caseSeriesIdentities). A separate step + * after `test-paid-shards.ts --report`, so the paid runner's closure never + * imports the history tool. + * + * Usage: bun run scripts/eval-trial-series.ts + */ +import * as fs from 'node:fs'; +import * as path from 'node:path'; +import { caseSeriesIdentities } from './eval-flake-rank'; +import { formatTrialOutcomes, parseTrialOutcomes } from '../test/helpers/eval-store'; + +const ROOT = path.resolve(import.meta.dir, '..'); + +/** Rewrite the file with every record stamped; an invalid line fails the whole stamp. */ +export function stampTrialSeries(file: string, root = ROOT): number { + const { records, errors } = parseTrialOutcomes(fs.readFileSync(file, 'utf8')); + if (errors.length) throw new Error(`${file}: ${errors.join('; ')}`); + const identities = caseSeriesIdentities([...new Set(records.map(record => record.case))], root); + const stamped = records.map(record => ({ ...record, series_identity: identities[record.case] })); + fs.writeFileSync(file, formatTrialOutcomes(stamped)); + return stamped.length; +} + +if (import.meta.main) { + const file = process.argv[2]; + if (!file) { + console.error('usage: bun run scripts/eval-trial-series.ts '); + process.exit(2); + } + console.log(`[eval-trial-series] stamped ${stampTrialSeries(file)} record(s) in ${file}`); +} diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 4fdd84b8e..198f8aceb 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -1739,11 +1739,18 @@ export function buildRunManifest(opts: { ordinary.filter(files => !registeredFiles.has(files[0])))) { const lanes = registeredFiles.has(files[0]) ? longLanes : ordinarySlices; const laneKeys = (index: number) => [...allocations].filter(([, lane]) => lane === index + 1).map(([key]) => key); - let lane = -1; - for (let index = 0; index < lanes; index++) { - if (sharesPanel(laneKeys(index), files[0]!)) continue; - if (lane < 0 || loads[index] < loads[lane]) lane = index; - } + // A trial whose siblings already hold every long lane may use any + // ordinary lane: independent runners outrank long-lane ownership. + const pick = (limit: number) => { + let best = -1; + for (let index = 0; index < limit; index++) { + if (sharesPanel(laneKeys(index), files[0]!)) continue; + if (best < 0 || loads[index] < loads[best]) best = index; + } + return best; + }; + let lane = pick(lanes); + if (lane < 0) lane = pick(ordinarySlices); if (lane < 0) lane = loads.slice(0, lanes).indexOf(Math.min(...loads.slice(0, lanes))); allocations.set(files[0], lane + 1); loads[lane] += resolvePaidShardTimeoutMs(files, opts.timeoutMs); @@ -2500,21 +2507,7 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo const laterPanels = attempts.slice(1).flatMap(attempt => panelReports(manifest, artifacts.map(a => a.result), attempt) .filter(panel => panel.trials.length > 0)); - // Quarantine policy checks on census runs: the per-tier cap and entry expiry. - if (manifest.evalsAll) { - const tierIds = Object.keys(E2E_TIERS).filter(id => E2E_TIERS[id] === manifest.tier); - const quarantined = Object.keys(CASE_QUARANTINE).filter(id => E2E_TIERS[id] === manifest.tier); - if (quarantined.length > EVAL_POLICY.quarantine.capFraction * tierIds.length) { - verdict.problems.push(`QUARANTINE over cap: ${quarantined.length} of ${tierIds.length} ${manifest.tier} cases (cap ${Math.round(EVAL_POLICY.quarantine.capFraction * 100)}%)`); - } - const expiryMs = EVAL_POLICY.quarantine.expiryWeeklyRuns * 7 * 24 * 60 * 60 * 1000; - for (const id of quarantined) { - const entered = Date.parse(CASE_QUARANTINE[id]!.enteredAt); - if (!Number.isFinite(entered) || Date.now() - entered > expiryMs) { - verdict.problems.push(`QUARANTINE expired: ${id} (entered ${CASE_QUARANTINE[id]!.enteredAt}; entries expire after ${EVAL_POLICY.quarantine.expiryWeeklyRuns} weekly runs)`); - } - } - } + // Quarantine cap and expiry are the weekly pass-rates gate's (eval-flake-rank --gate). // History: one trial-outcomes line per isolated trial and per JUnit rule/judge case. const runId = env.GITHUB_RUN_ID; @@ -2567,6 +2560,7 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo } } } + // series_identity is stamped afterwards by scripts/eval-trial-series.ts (the report job's next step). fs.writeFileSync(trialOutcomesPath, formatTrialOutcomes(history)); // Headline and failure block (A4): one formatter for the log, the PR comment and the weekly issue. diff --git a/test/ci-paid-coordination.test.ts b/test/ci-paid-coordination.test.ts index 7405037e9..5601e9087 100644 --- a/test/ci-paid-coordination.test.ts +++ b/test/ci-paid-coordination.test.ts @@ -3,7 +3,7 @@ import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; import { spawnSync } from 'node:child_process'; -import { buildRunManifest, collectPaidTestFiles, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards'; +import { buildRunManifest, collectPaidTestFiles, shardCaseId, shardTrial, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards'; import { STRICT_RETRY_CASE_BUDGETS } from './helpers/eval-budgets'; import { approvedCookieWorkflowSource, manualReviewFixture } from './helpers/manual-judge-review-fixture'; @@ -16,6 +16,11 @@ type Job = { permissions: Record; steps: Step[]; }; +/** A passing trial record for an isolated trial shard (the executor's current result schema). */ +const trialRecord = (entry: PaidRunManifest['entries'][number]) => entry.trial ? { trial: { + case: shardCaseId(entry.file)!, trial: shardTrial(entry.file)!, ...entry.trial, outcome: 'passed' as const, cost_usd: 0, duration_ms: 1, +} } : {}; + const workflows = ['evals.yml', 'evals-periodic.yml'].map(name => ({ name, jobs: (Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows', name), 'utf8')) as { @@ -217,6 +222,7 @@ describe('dependency-free CI planner and report execution', () => { executedTests: STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === entry.file)?.cases ?? 1, skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}), + ...trialRecord(entry), })), }; fs.writeFileSync(path.join(reportDir, `slice-${sliceIndex}.json`), JSON.stringify(result)); @@ -275,7 +281,7 @@ describe('dependency-free CI planner and report execution', () => { outcomes: manifest.entries.filter(entry => entry.status === 'planned').map(entry => ({ files: [entry.file], status: 'passed', exitCode: 0, elapsedMs: 1, executedTests: STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === entry.file)?.cases ?? 1, - skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}), + skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}), ...trialRecord(entry), })), }; const slicePath = path.join(reportDir, 'slice-1.json'); diff --git a/test/eng-finding-retry-budget.test.ts b/test/eng-finding-retry-budget.test.ts index a865fd8a3..b8aa70b78 100644 --- a/test/eng-finding-retry-budget.test.ts +++ b/test/eng-finding-retry-budget.test.ts @@ -1,5 +1,5 @@ import { expect, test } from 'bun:test'; -import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards'; +import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, expandTrialShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards'; import { FINDING_RETRY_BUDGETS, ALL_TIERS, SHARD_RESERVE_MS } from './helpers/eval-budgets'; import fs from 'node:fs'; import os from 'node:os'; @@ -175,8 +175,9 @@ test('single-slice manifest retains all registered files with one allocation', ( test('current detach supervision covers the live-census floor', () => { const floorFor = (tier: 'gate' | 'periodic') => { - // Case-sharded files contribute one shard per case, exactly as the runner plans. - const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier); + // Case-sharded files contribute one shard per case and isolated cases one + // shard per trial, exactly as the runner plans. + const files = expandTrialShards(expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier), tier).keys; const excess = files.reduce((n, file) => n + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - DEFAULT_SHARD_TIMEOUT_MS), 0); return Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05); }; @@ -186,7 +187,8 @@ test('current detach supervision covers the live-census floor', () => { expect(floorFor('gate')).toBe(21_725); expect(gateTimeout).toBe(49_320); expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate')); - expect(floorFor('periodic')).toBe(22_481); + expect(floorFor('periodic')).toBe(33_821); + expect(periodicTimeout).toBeGreaterThanOrEqual(floorFor('periodic')); }); for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => { diff --git a/test/helpers/eval-store.ts b/test/helpers/eval-store.ts index 32de29324..23ec90ffd 100644 --- a/test/helpers/eval-store.ts +++ b/test/helpers/eval-store.ts @@ -402,6 +402,8 @@ export interface TrialOutcomeRecord { sha?: string; lane?: string; recorded_at?: string; + /** History series key: a hash of the case's own touchfiles (GLOBAL_TOUCHFILES excluded), stamped by the report job. */ + series_identity?: string; } /** First line of free text, stripped of @-mentions and control characters, capped. */ @@ -434,6 +436,7 @@ function trialRecordProblems(r: any): string[] { if (r.execution !== 'executed' && r.execution !== 'reused') problems.push('execution invalid'); if (!['shard', 'junit', 'backfill'].includes(r.source)) problems.push('source invalid'); if (r.error !== undefined && (typeof r.error !== 'string' || r.error.length > TRIAL_ERROR_MAX)) problems.push('error invalid'); + if (r.series_identity !== undefined && (typeof r.series_identity !== 'string' || !/^[\w.-]{1,64}$/.test(r.series_identity))) problems.push('series_identity invalid'); return problems; } diff --git a/test/paid-free-boundary.test.ts b/test/paid-free-boundary.test.ts index 71a5350d3..45d9142ff 100644 --- a/test/paid-free-boundary.test.ts +++ b/test/paid-free-boundary.test.ts @@ -27,7 +27,7 @@ function runnerDependencies(root: string, entries: string[]): string[] { const source = fs.readFileSync(file, 'utf8').replace(/^#![^\n]*(?:\n|$)/, '\n'); let audited = source; if (relative === 'test/helpers/test-selection.ts') { - if (createHash('sha256').update(source).digest('hex') !== '4d2fbcb6249e8d22453d25bfe9b18ee0f4568bbec071918675c38a455d4e1e08') { + if (createHash('sha256').update(source).digest('hex') !== '052ad5a52472bcb41db04c9f21fe6a819e9547768468f7e5d390e0014b567677') { throw new Error('Re-audit the historical touchfile map loader before excluding its computed import'); } audited = source.replace('`const m = await import(${JSON.stringify(dataPath)});`,', "'',"); diff --git a/test/paid-report-fail-open.test.ts b/test/paid-report-fail-open.test.ts index a286c737c..27343310c 100644 --- a/test/paid-report-fail-open.test.ts +++ b/test/paid-report-fail-open.test.ts @@ -11,6 +11,7 @@ import * as os from 'node:os'; import * as path from 'node:path'; import { spawnSync } from 'node:child_process'; import { parseRunManifest, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards'; +import { stampTrialSeries } from '../scripts/eval-trial-series'; const ROOT = path.resolve(import.meta.dir, '..'); const RULE_A = 'test/skill-e2e-fail-open-alpha.test.ts'; @@ -148,8 +149,12 @@ describe('behavior and quarantined panels through --report', () => { const summary = JSON.parse(fs.readFileSync(path.join(r.dir, 'collector-outcomes.json'), 'utf8')); expect(summary.version).toBe(2); expect(summary.panels[0]).toMatchObject({ case: ID, status: 'PASS', split: true, failsLane: false }); - const history = fs.readFileSync(path.join(r.dir, 'trial-outcomes.jsonl'), 'utf8').trim().split('\n').map(line => JSON.parse(line)); - expect(history.map(h => [h.trial, h.outcome])).toEqual([[1, 'passed'], [2, 'failed'], [3, 'passed']]); + const outcomesFile = path.join(r.dir, 'trial-outcomes.jsonl'); + const history = () => fs.readFileSync(outcomesFile, 'utf8').trim().split('\n').map(line => JSON.parse(line)); + expect(history().map(h => [h.trial, h.outcome])).toEqual([[1, 'passed'], [2, 'failed'], [3, 'passed']]); + expect(stampTrialSeries(outcomesFile)).toBe(3); + expect(new Set(history().map(h => h.series_identity)).size).toBe(1); + expect(history()[0].series_identity).toMatch(/^[0-9a-f]{16}$/); }); test('behavior 1/3: red', () => { diff --git a/test/paid-retry-supervision.test.ts b/test/paid-retry-supervision.test.ts index cb989a9c2..7f0e93796 100644 --- a/test/paid-retry-supervision.test.ts +++ b/test/paid-retry-supervision.test.ts @@ -230,8 +230,8 @@ test('detached PR fallback and release commands cover their actual default worke )) / 1000 * 1.05)); } const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000; - expect(releaseFloors).toEqual([21_725, 22_481]); - expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(44_206); + expect(releaseFloors).toEqual([21_725, 33_821]); + expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(55_546); expect(detachedReleaseWall).toBe(116_700_000); expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000); });