fix(evals): tsx-safe generics in eval-flake-rank, legacy artifact names, no-retry wall docs

This commit is contained in:
garrytan committed 2026-09-29 19:47:44 +00:00
1 parent 32772f535d
commit 1e640bbfb7
3 files changed
+22 -19

No files matched your search

+14 -13
View File
@@ -492,22 +492,23 @@ minus overhead and ratchets raw literals. Budget above the wall is fiction.
No paid test may exceed the ordinary tiers.
`FINDING_RETRY_BUDGETS` also registers the CEO split-overflow and Eng
multi-finding batching files. Each retains its 25-minute case deadline and, as a
case past the retry cap, runs once in a 27-minute shard wall including two
minutes for cleanup. No per-case budget grows. Overlay wrappers
multi-finding batching files. Each retains its 25-minute case deadline and runs
once (paid evals never retry) in a 27-minute shard wall including two minutes for
cleanup. No per-case budget grows. Overlay wrappers
have a 1,830-second minimum shard wall and run without Bun retries; see the
[overlay contract](OVERLAY_BENCHMARK_CONTRACT.md) for their unchanged work budget.
The quality file reserves its whole-file wall for every case and its one retry
(judge cases are under the retry cap), plus cleanup. Each still has 120 seconds of model work. Its 17 workflow
The quality file reserves its whole-file wall (3,170 seconds) for every case run
once, plus cleanup. Each still has 120 seconds of model work. Its 17 workflow
judges own their deadline and abort signal, with five seconds for terminal
recording inside a ten-second Bun grace; the other 11 retain their existing
120-second Bun timeout. Late responses cannot create records or cache passes.
The ship documentation file reserves 5,520 seconds for five 600-second cases and
eight 300-second fault cases, run once (a 600-second case is past the retry cap), plus cleanup. The standalone
The ship documentation file reserves 4,920 seconds for four 600-second cases and
eight 300-second fault cases, run once, plus cleanup; in CI each case runs as its
own shard. The standalone
documentation child retains its 600-second case. The five review/ship explorer
cases reserve 3,270 seconds including their existing retry and finalization grace.
cases reserve 1,695 seconds, run once, including finalization grace.
These are whole-file supervision limits, not additional model work per case.
The shared-library path file reserves 1,920 seconds for its three serial
@@ -521,7 +522,7 @@ records fail reconciliation. Case deadlines and model budgets do not grow.
`resolvePaidShardBudget(files, overrideMs?)` is the canonical per-job resolver.
Each registered finding file and each overlay wrapper requires its
own shard, even with `--files-per-shard` above one. Mixed or multi-file overlay
jobs are rejected so ordinary files retain their configured retries. An explicit
jobs are rejected. An explicit
CLI `--timeout`, `EVALS_SHARD_TIMEOUT_MS`, or API `timeoutMs` still wins for these
policies, including a lower cap; overlay overrides below their minimum are rejected.
Planner entries and execution results record the effective wall,
@@ -529,13 +530,13 @@ its source and policy identifier. Custom drivers must resolve each job instead
of passing their ordinary 1800-second default as an explicit cap;
their outer controller/detach wall must also cover the allocated work and cleanup.
The paid census counts are printed by `--list` for each tier.
`eval:bg:pr` and `eval:bg:periodic` have 92820/67380-second outer caps; the PR
`eval:bg:pr` and `eval:bg:periodic` have 92820/67380-second outer caps, above their recomputed floors (PR fallback 72,755 s, periodic 33,821 s including the trial shards); the PR
wrapper covers a full-gate fallback at its default two workers. The broad gate
wrapper reserves 49320 seconds, and release reserves 116700 seconds for both
wrapper reserves 49320 seconds (floor 21,725 s), and release reserves 116700 seconds for both
tiers; free tests recompute each floor from the live shard census, case shards
included. Legacy monolithic
`eval:bg`/`eval:bg:all` retain their shorter 5400/7200-second caps and do not
promise every registered retry; use the sharded periodic path for this policy.
`eval:bg`/`eval:bg:all` retain their shorter 5400/7200-second caps; use the
sharded periodic path for complete coverage.
CI plans with `--slice-budget 540 --jobs 2` for the PR gate, the periodic census
and the weekly gate census (the gate census also `--skip-judges`), and
+6 -4
View File
@@ -578,13 +578,15 @@ function gh(args: string[]): Buffer {
return result.stdout;
}
const jsonLines = <T>(buffer: Buffer): T[] => buffer.toString('utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as T);
function jsonLines<T>(buffer: Buffer): T[] {
return buffer.toString('utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as T);
}
export const GH_HISTORY: HistoryFetcher = {
listRuns: (repo, workflow, branch, limit) => jsonLines<WeeklyRun>(gh(['api',
listRuns: (repo, workflow, branch, limit): WeeklyRun[] => jsonLines(gh(['api',
`repos/${repo}/actions/workflows/${workflow}/runs?branch=${encodeURIComponent(branch)}&status=completed&per_page=${limit}`,
'--jq', '.workflow_runs[] | {id, attempt: .run_attempt, sha: .head_sha, branch: .head_branch, createdAt: .created_at}'])),
listArtifacts: (repo, runId) => jsonLines<RunArtifact>(gh(['api', `repos/${repo}/actions/runs/${runId}/artifacts?per_page=100`,
listArtifacts: (repo, runId): RunArtifact[] => jsonLines(gh(['api', `repos/${repo}/actions/runs/${runId}/artifacts?per_page=100`,
'--paginate', '--jq', '.artifacts[] | select(.expired | not) | {id, name, size: .size_in_bytes}'])),
downloadZip: (repo, artifactId, destination) => fs.writeFileSync(destination, gh(['api', `repos/${repo}/actions/artifacts/${artifactId}/zip`])),
};
@@ -674,7 +676,7 @@ if (import.meta.main) {
weeklyRuns = runs.map(run => run.createdAt);
const cacheDir = path.join(os.homedir(), '.gstack', 'eval-pass-rates-cache', repo.replace('/', '-'));
const match = backfill
? (name: string) => name.startsWith('trial-outcomes') || /^(paid-slice-\d+|gate-census-\d+)$/.test(name)
? (name: string) => name.startsWith('trial-outcomes') || /^(paid-slice-\d+|gate-census-\d+)(-a\d+)?$/.test(name)
: (name: string) => name.startsWith('trial-outcomes');
for (const run of runs) {
const dirsForRun = downloadRunArtifacts({ repo, run, match, cacheDir, maxBytes: backfill ? 64 * 1024 * 1024 : undefined });
+2 -2
View File
@@ -270,8 +270,8 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
// Covers ceo (preamble misfire) + eng/design (scope-gate bypass must not
// fire outside plan mode) + the named-target exception case. 4 PTY runs;
// in CI these run CONCURRENT with the rest of the pty-plan-smoke suite
// (--max-concurrency + --retry 1), so worst-case cost is ~2x a single
// pass of each, sharing the API budget with sibling tests — not the
// (--max-concurrency, no retries), so worst-case cost is one pass of
// each, sharing the API budget with sibling tests — not the
// sequential ~+10min a local read suggests.
'plan-mode-no-op': [