From 1e640bbfb7c368c0495bd95e312035938248a6ab Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:47:44 +0000 Subject: [PATCH] fix(evals): tsx-safe generics in eval-flake-rank, legacy artifact names, no-retry wall docs --- docs/TESTING_INTERNALS.md | 27 ++++++++++++++------------- scripts/eval-flake-rank.ts | 10 ++++++---- test/helpers/touchfiles-data.ts | 4 ++-- 3 files changed, 22 insertions(+), 19 deletions(-) diff --git a/docs/TESTING_INTERNALS.md b/docs/TESTING_INTERNALS.md index 0cb64ac28..af5b04398 100644 --- a/docs/TESTING_INTERNALS.md +++ b/docs/TESTING_INTERNALS.md @@ -492,22 +492,23 @@ minus overhead and ratchets raw literals. Budget above the wall is fiction. No paid test may exceed the ordinary tiers. `FINDING_RETRY_BUDGETS` also registers the CEO split-overflow and Eng -multi-finding batching files. Each retains its 25-minute case deadline and, as a -case past the retry cap, runs once in a 27-minute shard wall including two -minutes for cleanup. No per-case budget grows. Overlay wrappers +multi-finding batching files. Each retains its 25-minute case deadline and runs +once (paid evals never retry) in a 27-minute shard wall including two minutes for +cleanup. No per-case budget grows. Overlay wrappers have a 1,830-second minimum shard wall and run without Bun retries; see the [overlay contract](OVERLAY_BENCHMARK_CONTRACT.md) for their unchanged work budget. -The quality file reserves its whole-file wall for every case and its one retry -(judge cases are under the retry cap), plus cleanup. Each still has 120 seconds of model work. Its 17 workflow +The quality file reserves its whole-file wall (3,170 seconds) for every case run +once, plus cleanup. Each still has 120 seconds of model work. Its 17 workflow judges own their deadline and abort signal, with five seconds for terminal recording inside a ten-second Bun grace; the other 11 retain their existing 120-second Bun timeout. Late responses cannot create records or cache passes. -The ship documentation file reserves 5,520 seconds for five 600-second cases and -eight 300-second fault cases, run once (a 600-second case is past the retry cap), plus cleanup. The standalone +The ship documentation file reserves 4,920 seconds for four 600-second cases and +eight 300-second fault cases, run once, plus cleanup; in CI each case runs as its +own shard. The standalone documentation child retains its 600-second case. The five review/ship explorer -cases reserve 3,270 seconds including their existing retry and finalization grace. +cases reserve 1,695 seconds, run once, including finalization grace. These are whole-file supervision limits, not additional model work per case. The shared-library path file reserves 1,920 seconds for its three serial @@ -521,7 +522,7 @@ records fail reconciliation. Case deadlines and model budgets do not grow. `resolvePaidShardBudget(files, overrideMs?)` is the canonical per-job resolver. Each registered finding file and each overlay wrapper requires its own shard, even with `--files-per-shard` above one. Mixed or multi-file overlay -jobs are rejected so ordinary files retain their configured retries. An explicit +jobs are rejected. An explicit CLI `--timeout`, `EVALS_SHARD_TIMEOUT_MS`, or API `timeoutMs` still wins for these policies, including a lower cap; overlay overrides below their minimum are rejected. Planner entries and execution results record the effective wall, @@ -529,13 +530,13 @@ its source and policy identifier. Custom drivers must resolve each job instead of passing their ordinary 1800-second default as an explicit cap; their outer controller/detach wall must also cover the allocated work and cleanup. The paid census counts are printed by `--list` for each tier. -`eval:bg:pr` and `eval:bg:periodic` have 92820/67380-second outer caps; the PR +`eval:bg:pr` and `eval:bg:periodic` have 92820/67380-second outer caps, above their recomputed floors (PR fallback 72,755 s, periodic 33,821 s including the trial shards); the PR wrapper covers a full-gate fallback at its default two workers. The broad gate -wrapper reserves 49320 seconds, and release reserves 116700 seconds for both +wrapper reserves 49320 seconds (floor 21,725 s), and release reserves 116700 seconds for both tiers; free tests recompute each floor from the live shard census, case shards included. Legacy monolithic -`eval:bg`/`eval:bg:all` retain their shorter 5400/7200-second caps and do not -promise every registered retry; use the sharded periodic path for this policy. +`eval:bg`/`eval:bg:all` retain their shorter 5400/7200-second caps; use the +sharded periodic path for complete coverage. CI plans with `--slice-budget 540 --jobs 2` for the PR gate, the periodic census and the weekly gate census (the gate census also `--skip-judges`), and diff --git a/scripts/eval-flake-rank.ts b/scripts/eval-flake-rank.ts index 363fd66d8..5873d5e29 100644 --- a/scripts/eval-flake-rank.ts +++ b/scripts/eval-flake-rank.ts @@ -578,13 +578,15 @@ function gh(args: string[]): Buffer { return result.stdout; } -const jsonLines = (buffer: Buffer): T[] => buffer.toString('utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as T); +function jsonLines(buffer: Buffer): T[] { + return buffer.toString('utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as T); +} export const GH_HISTORY: HistoryFetcher = { - listRuns: (repo, workflow, branch, limit) => jsonLines(gh(['api', + listRuns: (repo, workflow, branch, limit): WeeklyRun[] => jsonLines(gh(['api', `repos/${repo}/actions/workflows/${workflow}/runs?branch=${encodeURIComponent(branch)}&status=completed&per_page=${limit}`, '--jq', '.workflow_runs[] | {id, attempt: .run_attempt, sha: .head_sha, branch: .head_branch, createdAt: .created_at}'])), - listArtifacts: (repo, runId) => jsonLines(gh(['api', `repos/${repo}/actions/runs/${runId}/artifacts?per_page=100`, + listArtifacts: (repo, runId): RunArtifact[] => jsonLines(gh(['api', `repos/${repo}/actions/runs/${runId}/artifacts?per_page=100`, '--paginate', '--jq', '.artifacts[] | select(.expired | not) | {id, name, size: .size_in_bytes}'])), downloadZip: (repo, artifactId, destination) => fs.writeFileSync(destination, gh(['api', `repos/${repo}/actions/artifacts/${artifactId}/zip`])), }; @@ -674,7 +676,7 @@ if (import.meta.main) { weeklyRuns = runs.map(run => run.createdAt); const cacheDir = path.join(os.homedir(), '.gstack', 'eval-pass-rates-cache', repo.replace('/', '-')); const match = backfill - ? (name: string) => name.startsWith('trial-outcomes') || /^(paid-slice-\d+|gate-census-\d+)$/.test(name) + ? (name: string) => name.startsWith('trial-outcomes') || /^(paid-slice-\d+|gate-census-\d+)(-a\d+)?$/.test(name) : (name: string) => name.startsWith('trial-outcomes'); for (const run of runs) { const dirsForRun = downloadRunArtifacts({ repo, run, match, cacheDir, maxBytes: backfill ? 64 * 1024 * 1024 : undefined }); diff --git a/test/helpers/touchfiles-data.ts b/test/helpers/touchfiles-data.ts index 860afc9f2..4af7e14d6 100644 --- a/test/helpers/touchfiles-data.ts +++ b/test/helpers/touchfiles-data.ts @@ -270,8 +270,8 @@ export const E2E_TOUCHFILES: Record = { // Covers ceo (preamble misfire) + eng/design (scope-gate bypass must not // fire outside plan mode) + the named-target exception case. 4 PTY runs; // in CI these run CONCURRENT with the rest of the pty-plan-smoke suite - // (--max-concurrency + --retry 1), so worst-case cost is ~2x a single - // pass of each, sharing the API budget with sibling tests — not the + // (--max-concurrency, no retries), so worst-case cost is one pass of + // each, sharing the API budget with sibling tests — not the // sequential ~+10min a local read suggests. 'plan-mode-no-op': [