From b1f5bc0032cf694d5ae4167bd3a1eb99a6046350 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:09:51 +0000 Subject: [PATCH] test(evals): retire every paid automatic retry Paid evals never retry (approved 2026-09-29): delete SHORT_CASE_RETRY_FILES and retriesWithinCaseCap, drop the retry fields from the registered wall rows (walls now cover one run plus reserve), make retriesForFiles return 0, pass --retry 0 explicitly, and drop --retry 1 from the package.json paid scripts. Add the eval:pass-rates alias. Tests that pinned the old retry allowance are updated as a policy change; review-finalization-budget now proves late-result recording under the production zero-retry arguments. --- package.json | 13 +- scripts/test-paid-shards.ts | 31 ++--- test/carve-section-sharding.test.ts | 2 +- test/eng-finding-retry-budget.test.ts | 13 +- test/helpers/eval-budgets.ts | 148 ++++++++--------------- test/paid-overlay-scheduling.test.ts | 12 +- test/paid-retry-supervision.test.ts | 76 +++++------- test/paid-run-manifest.test.ts | 13 +- test/review-finalization-budget.test.ts | 20 ++- test/ship-hook-actor.test.ts | 4 +- test/ship-skip-actor.test.ts | 4 +- test/workflow-boundaries-fixture.test.ts | 4 +- 12 files changed, 131 insertions(+), 209 deletions(-) diff --git a/package.json b/package.json index fb8002dbd..86ddbcbc1 100644 --- a/package.json +++ b/package.json @@ -31,12 +31,12 @@ "test:free": "bun run scripts/test-free-shards.ts", "test:windows": "bun run scripts/test-free-shards.ts --windows-only", "test:ubicloud": "bash scripts/ubicloud/test-free.sh", - "test:evals": "EVALS=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", - "test:evals:all": "EVALS=1 EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", - "test:e2e": "EVALS=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts", - "test:e2e:all": "EVALS=1 EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts", - "test:gate": "EVALS=1 EVALS_TIER=gate bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", - "test:periodic": "EVALS=1 EVALS_TIER=periodic EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:evals": "EVALS=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:evals:all": "EVALS=1 EVALS_ALL=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:e2e": "EVALS=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts", + "test:e2e:all": "EVALS=1 EVALS_ALL=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts", + "test:gate": "EVALS=1 EVALS_TIER=gate bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:periodic": "EVALS=1 EVALS_TIER=periodic EVALS_ALL=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", "test:gate:sharded": "bun run scripts/test-paid-shards.ts --tier gate", "test:periodic:sharded": "EVALS_ALL=1 bun run scripts/test-paid-shards.ts --tier periodic", "test:codex": "EVALS=1 bun test test/codex-e2e.test.ts test/codex-e2e-sol-scope.test.ts", @@ -52,6 +52,7 @@ "eval:compare": "bun run scripts/eval-compare.ts", "eval:summary": "bun run scripts/eval-summary.ts", "eval:flake-rank": "bun run scripts/eval-flake-rank.ts", + "eval:pass-rates": "bun run scripts/eval-flake-rank.ts", "eval:watch": "bun run scripts/eval-watch.ts", "eval:select": "bun run scripts/eval-select.ts", "analytics": "bun run scripts/analytics.ts", diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 8790ac71d..c4f7031b8 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -67,7 +67,7 @@ import { } from './test-strict-output'; import { PAID_TEST_GLOBS, isPaidTestFile } from '../test/helpers/paid-test-set'; import { CASE_CI_EXCLUDE, PERIODIC_CI_EXCLUDE } from '../test/helpers/periodic-exclude-data'; -import { FILE_RETRY_BUDGETS, SHORT_CASE_RETRY_FILES, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets'; +import { FILE_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets'; import { getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile, evalEntryOutcome } from '../test/helpers/eval-store'; import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-review'; import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight'; @@ -644,9 +644,9 @@ export function resolvePaidShardBudget(files: string[], overrideMs?: number): Pa if (overlay && overrideMs !== undefined && overrideMs < OVERLAY_MIN_FILE_WALL_MS) { throw new Error(`Overlay shard requires at least ${OVERLAY_MIN_FILE_WALL_MS}ms; explicit wall ${overrideMs}ms cannot preserve its work and finalization budget`); } - // A registered file's case shard supervises one case and its allowed attempts. + // A registered file's case shard supervises its one case. const registeredMs = finding && shardCaseId(files[0]!) !== null - ? finding.caseMs * (finding.retries + 1) + finding.shardReserveMs : finding?.shardMs; + ? finding.caseMs + finding.shardReserveMs : finding?.shardMs; return { timeoutMs: overrideMs ?? (registeredMs ?? (overlay ? OVERLAY_MIN_FILE_WALL_MS : DEFAULT_SHARD_TIMEOUT_MS)), source: overrideMs !== undefined ? 'explicit' : finding ? 'registered' : 'default', @@ -667,9 +667,9 @@ export function buildPaidShardArgs( // Explicit --concurrent/--max-concurrency: the legacy path always set one; // omitting it here made within-shard parallelism differ silently between // the two runners (observed: 1.6x sumdur/wall sharded vs 8x legacy). - // Retries come from retriesForFiles (the timeout-is-a-verdict rule) at the - // call site; the fallback of 1 serves only direct callers. - return ['test', ...files, '--retry', String(retries ?? 1), '--concurrent', `--max-concurrency=${maxConcurrency}`, `--timeout=${timeoutMs}`]; + // Paid evals never retry (retriesForFiles); `--retry 0` is explicit so a + // bunfig default can never reintroduce one. + return ['test', ...files, '--retry', String(retries ?? 0), '--concurrent', `--max-concurrency=${maxConcurrency}`, `--timeout=${timeoutMs}`]; } /** @@ -1254,20 +1254,13 @@ export interface PaidRunManifest { } /** - * Automatic retries follow the approved rule in test/helpers/eval-budgets.ts: - * a timed-out attempt is a verdict, so only files whose every case budget is - * at most RETRY_MAX_CASE_MS keep a retry (registered rows derive it from their - * caseMs; SHORT_CASE_RETRY_FILES lists the rest). Overlays and every other - * paid file run once. A multi-file shard takes the smallest allowance. + * Paid evals never retry (approved 2026-09-29): a failed verdict is final for + * its run, and trials are fixed by kind before the run (EVAL_POLICY). The + * function stays the single statement of that policy for the Bun arguments + * and the reuse identity. */ -export function retriesForFiles(files: string[]): number { - if (files.some(isOverlayTestFile)) return 0; - return Math.min(...files.map((file) => { - const rel = shardFile(file); - const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === rel); - if (registered) return registered.retries; - return SHORT_CASE_RETRY_FILES.includes(rel) ? 1 : 0; - })); +export function retriesForFiles(_files: string[]): number { + return 0; } export const PAID_TEST_DURATIONS_FILE = 'scripts/paid-test-durations.json'; diff --git a/test/carve-section-sharding.test.ts b/test/carve-section-sharding.test.ts index 4f60cfa43..e5710c166 100644 --- a/test/carve-section-sharding.test.ts +++ b/test/carve-section-sharding.test.ts @@ -22,7 +22,7 @@ describe('carved-skill cases each get a complete paid process budget', () => { expect(selectPaidTestFiles(files.map(file => 'test/' + file), 'periodic').selected).toHaveLength(files.length); expect(selectPaidTestFiles(files.map(file => 'test/' + file), 'gate').selected).toHaveLength(0); }); - test('all configured retries plus teardown fit even with within-shard concurrency one', () => { + test('every case run plus teardown fits even with within-shard concurrency one', () => { for (const file of files) { const attempts = retriesForFiles(['test/' + file]) + 1; expect(CAPTURE_LONG_MS * attempts + 10_000).toBeLessThan(DEFAULT_SHARD_TIMEOUT_MS); diff --git a/test/eng-finding-retry-budget.test.ts b/test/eng-finding-retry-budget.test.ts index 4b313e03b..a865fd8a3 100644 --- a/test/eng-finding-retry-budget.test.ts +++ b/test/eng-finding-retry-budget.test.ts @@ -6,13 +6,12 @@ import os from 'node:os'; import path from 'node:path'; for (const budget of FINDING_RETRY_BUDGETS) { - test(`${budget.file}: supervision preserves every existing attempt and retry`, () => { + test(`${budget.file}: supervision covers its one run of every case`, () => { expect(budget.testMs).toBe(1_500_000); - // A 25-minute case is past RETRY_MAX_CASE_MS: a timed-out attempt is its verdict. - expect(budget.retries).toBe(0); - expect(retriesForFiles([budget.file])).toBe(budget.retries); + // Paid evals never retry: a timed-out case is its verdict. + expect(retriesForFiles([budget.file])).toBe(0); expect(budget.shardReserveMs).toBe(SHARD_RESERVE_MS); - expect(budget.shardMs).toBe(budget.cases * budget.testMs * (budget.retries + 1) + budget.shardReserveMs); + expect(budget.shardMs).toBe(budget.cases * budget.testMs + budget.shardReserveMs); expect(resolvePaidShardBudget([budget.file])).toEqual({ timeoutMs: budget.shardMs, source: 'registered', policyId: budget.id }); const source = fs.readFileSync(path.join(import.meta.dir, '..', budget.file), 'utf8'); if (budget.file === 'test/skill-e2e-plan-ceo-split-overflow.test.ts') { @@ -184,10 +183,10 @@ test('current detach supervision covers the live-census floor', () => { const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8')); const periodicTimeout = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]); const gateTimeout = Number(pkg.scripts['eval:bg:gate'].match(/--timeout\s+(\d+)/)[1]); - expect(floorFor('gate')).toBe(26_471); + expect(floorFor('gate')).toBe(21_725); expect(gateTimeout).toBe(49_320); expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate')); - expect(floorFor('periodic')).toBe(30_797); + expect(floorFor('periodic')).toBe(22_481); }); for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => { diff --git a/test/helpers/eval-budgets.ts b/test/helpers/eval-budgets.ts index 023e6e4ac..a195c6f99 100644 --- a/test/helpers/eval-budgets.ts +++ b/test/helpers/eval-budgets.ts @@ -47,132 +47,80 @@ export const ALL_TIERS = { export const SHARD_RESERVE_MS = 2 * 60_000; /** - * Retry policy (approved 2026-09-29): a timed-out attempt is a verdict. Bun's - * --retry reruns a failed case after it may have spent its whole budget, so an - * automatic retry is kept only where one more attempt is short: every case of - * the file has a per-attempt budget of at most RETRY_MAX_CASE_MS, the CAPTURE - * tier plus its recording grace. Those failures are fast flake classes (API - * blips, tool hiccups) and a retry costs at most one more short attempt. Files - * with any longer case run once. Per-case budgets never change with this rule. + * Retry policy (approved 2026-09-29, eval reliability wave): paid evals never + * retry. Each case's kind (E2E_KINDS) fixes its trials before the run: `rule` + * one trial, `behavior` a panel of EVAL_POLICY.panel independent trials, and + * `judge` one case that samples its judge panel internally. A failed verdict + * is final for that run; a manual re-run adds trials under a new run attempt + * and never replaces the original verdict. Rows below keep only wall + * supervision; per-case budgets never change with this rule. */ -export const RETRY_MAX_CASE_MS = CAPTURE_MS + 15_000; -export function retriesWithinCaseCap(caseMs: number, configuredRetries: number): number { - return caseMs <= RETRY_MAX_CASE_MS ? configuredRetries : 0; -} - -/** - * Unregistered paid files that keep one automatic retry: every case budget is - * JUDGE or CAPTURE tier (test/paid-retry-supervision.test.ts scans each source). - * Registered rows below derive retries from their declared caseMs; every other - * paid file runs once. - */ -export const SHORT_CASE_RETRY_FILES: readonly string[] = [ - 'test/codex-e2e-sol-scope.test.ts', - 'test/llm-judge-recommendation.test.ts', - 'test/skill-e2e-ask-user-question-format-compliance.test.ts', - 'test/skill-e2e-benchmark-providers.test.ts', - 'test/skill-e2e-bws.test.ts', - 'test/skill-e2e-context-skills.test.ts', - 'test/skill-e2e-coverage-audit.test.ts', - 'test/skill-e2e-diagram.test.ts', - 'test/skill-e2e-first-task-scaffold.test.ts', - 'test/skill-e2e-gbrain-roundtrip-local.test.ts', - 'test/skill-e2e-hermetic-canary.test.ts', - 'test/skill-e2e-investigate-owned-completion.test.ts', - 'test/skill-e2e-investigate-owned-termination.test.ts', - 'test/skill-e2e-learnings.test.ts', - 'test/skill-e2e-plan-tune.test.ts', - 'test/skill-e2e-qa-functional-fix.test.ts', - 'test/skill-e2e-qa-functional.test.ts', - 'test/skill-e2e-review-army.test.ts', - 'test/skill-e2e-review.test.ts', - 'test/skill-e2e-session-intelligence.test.ts', - 'test/skill-e2e-setup-gbrain-bad-token.test.ts', - 'test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts', - 'test/skill-e2e-setup-gbrain-remote.test.ts', - 'test/skill-e2e-ship-hook-consent.test.ts', - 'test/skill-e2e-ship-hook-refresh.test.ts', - 'test/skill-e2e-ship-skip.test.ts', - 'test/skill-e2e-sync-gbrain-readiness.test.ts', - 'test/skill-e2e-third-party-actions.test.ts', - 'test/skill-e2e-triage.test.ts', - 'test/skill-routing-e2e.test.ts', -]; - -/** Whole-file supervision covers every attempt the retry policy allows. - * These fixtures allow 25 minutes per case, so they run once. +/** Whole-file supervision for one run of every case. + * These fixtures allow 25 minutes per case. * Reserve the sequential upper bound even when Bun runs sibling cases together. */ export const FINDING_RETRY_BUDGETS = [ { file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 }, { file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 }, -].map(({ file, cases }) => { - const retries = retriesWithinCaseCap(1_500_000, 1); - return { - file, cases, - id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`, - testMs: 1_500_000, - caseMs: 1_500_000, - retries, - shardReserveMs: SHARD_RESERVE_MS, - shardMs: cases * 1_500_000 * (retries + 1) + SHARD_RESERVE_MS, - }; -}); +].map(({ file, cases }) => ({ + file, cases, + id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`, + testMs: 1_500_000, + caseMs: 1_500_000, + shardReserveMs: SHARD_RESERVE_MS, + shardMs: cases * 1_500_000 + SHARD_RESERVE_MS, +})); -/** Three existing captures in one 16-minute case, so the file runs once. */ +/** Three existing captures in one 16-minute case. */ export const AUQ_CONSISTENCY_RETRY_BUDGET = { file: 'test/skill-e2e-auq-consistency.test.ts', id: 'auq-consistency-existing-retry-v1', cases: 1, testMs: 3 * CAPTURE_MS + 60_000, caseMs: 3 * CAPTURE_MS + 60_000, - retries: retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1), shardReserveMs: SHARD_RESERVE_MS, - shardMs: (3 * CAPTURE_MS + 60_000) * (retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1) + 1) + SHARD_RESERVE_MS, + shardMs: 3 * CAPTURE_MS + 60_000 + SHARD_RESERVE_MS, } as const; /** These fixtures have a fixed case count in every supported tier. */ export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET]; -/** Whole-file walls cover all existing cases and every allowed attempt, even if - * Bun runs them sequentially. Mixed-tier files reserve their larger complete - * tier, never a currently selected subset. caseMs is the longest single case - * budget, which decides the retry (RETRY_MAX_CASE_MS). These rows add no - * case-count or model-work policy. The 10-second terms preserve the existing - * Codex/recording finalization grace. +/** Whole-file walls cover all existing cases, even if Bun runs them + * sequentially. Mixed-tier files reserve their larger complete tier, never a + * currently selected subset. caseMs is the longest single case budget, the + * wall of one isolated case shard. These rows add no case-count or model-work + * policy. The 10-second terms preserve the existing Codex/recording + * finalization grace. */ export const FILE_RETRY_BUDGETS = [ ...STRICT_RETRY_CASE_BUDGETS, ...[ - { file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000, configuredRetries: 1 }, - { file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000 }, + { file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS }, + { file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS }, // Seventeen workflow judges include their 10s recording grace; the other - // seven judges retain 120s. Supervise all 24 and the existing one retry. - { file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000, configuredRetries: 1 }, - { file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 }, - { file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 2 }, - { file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, - { file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 }, + // seven judges retain 120s. Supervise all 24. + { file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000 }, + { file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS }, + { file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 }, + { file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS }, + { file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS }, + { file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS }, + { file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS }, + { file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS }, + { file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS }, + { file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS }, + { file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS }, + { file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 }, // Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s. - { file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS, configuredRetries: 1 }, - ].map(({ file, attemptMs, caseMs, configuredRetries }) => { - const retries = retriesWithinCaseCap(caseMs, configuredRetries); - return { - file, attemptMs, caseMs, retries, - id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`, - shardReserveMs: SHARD_RESERVE_MS, - shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS, - }; - }), + { file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS }, + ].map(({ file, attemptMs, caseMs }) => ({ + file, attemptMs, caseMs, + id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`, + shardReserveMs: SHARD_RESERVE_MS, + shardMs: attemptMs + SHARD_RESERVE_MS, + })), ]; /** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */ diff --git a/test/paid-overlay-scheduling.test.ts b/test/paid-overlay-scheduling.test.ts index 701eb5c07..7b08b466c 100644 --- a/test/paid-overlay-scheduling.test.ts +++ b/test/paid-overlay-scheduling.test.ts @@ -21,8 +21,8 @@ const fakeEnv = { }; describe('overlay file policy', () => { - test('grouped planning isolates every overlay and preserves ordinary retries', () => { - // Two short-case files keep their one retry (timeout-is-a-verdict rule). + test('grouped planning isolates every overlay and never retries ordinary files', () => { + // Paid evals never retry (approved 2026-09-29), short-case files included. const workflow = 'test/skill-e2e-review.test.ts'; const files = [...overlayFiles, 'test/skill-e2e-triage.test.ts', workflow]; for (const maxFilesPerShard of [2, 3, 10]) { @@ -31,9 +31,9 @@ describe('overlay file policy', () => { for (const file of overlayFiles) expect(shards).toContainEqual([file]); const workflowShard = shards.find(shard => shard.includes(workflow))!; expect(workflowShard.some(isOverlayTestFile)).toBe(false); - expect(retriesForFiles(workflowShard)).toBe(1); + expect(retriesForFiles(workflowShard)).toBe(0); const args = buildPaidShardArgs(workflowShard, resolvePaidShardTimeoutMs(workflowShard), 2, retriesForFiles(workflowShard)); - expect(args[args.indexOf('--retry') + 1]).toBe('1'); + expect(args[args.indexOf('--retry') + 1]).toBe('0'); expect(planPaidShards(files.map(file => file.replaceAll('/', '\\')), { maxFilesPerShard })).toEqual(shards); } }); @@ -66,10 +66,10 @@ describe('overlay file policy', () => { for (const file of [normalFile, 'test/skill-e2e-overlay-harness.test.ts', 'test/model-overlays.test.ts']) { expect(isOverlayTestFile(file)).toBe(false); expect(resolvePaidShardTimeoutMs([file])).toBe(DEFAULT_SHARD_TIMEOUT_MS); - // Not overlays; unlisted files run once because their case budget is unknown. + // Not overlays; every paid file runs once. expect(retriesForFiles([file])).toBe(0); } - expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1); + expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(0); expect(resolvePaidShardTimeoutMs([normalFile], 1234)).toBe(1234); expect(resolvePaidShardTimeoutMs([overlayFiles[0]], 1_900_000)).toBe(1_900_000); expect(() => resolvePaidShardTimeoutMs([overlayFiles[0]], 1_800_000)).toThrow('explicit wall'); diff --git a/test/paid-retry-supervision.test.ts b/test/paid-retry-supervision.test.ts index 4cf8eb92a..cb989a9c2 100644 --- a/test/paid-retry-supervision.test.ts +++ b/test/paid-retry-supervision.test.ts @@ -7,7 +7,7 @@ import { shardFile, sliceExecutionOrder, sliceSupervisedWallMs, CASE_SHARDED_FILES, } from '../scripts/test-paid-shards'; import { - ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, RETRY_MAX_CASE_MS, SHORT_CASE_RETRY_FILES, + ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, FINDING_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS, } from './helpers/eval-budgets'; @@ -15,16 +15,16 @@ import { E2E_TOUCHFILES } from './helpers/touchfiles'; const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8'); const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file)); -// Walls cover every attempt the retry rule allows: files with a case budget -// past RETRY_MAX_CASE_MS run once (a timed-out attempt is a verdict). +// Paid evals never retry (approved 2026-09-29): each wall covers one run of +// every case plus the supervision reserve. const expectedWalls = { - 'test/skill-e2e-qa-callers.test.ts': 3_270_000, + 'test/skill-e2e-qa-callers.test.ts': 1_695_000, 'test/skill-e2e-shared-libs-paths.test.ts': 1_920_000, 'test/skill-e2e-ship-docsync.test.ts': 4_920_000, - 'test/skill-llm-eval.test.ts': 6_220_000, + 'test/skill-llm-eval.test.ts': 3_170_000, 'test/skill-e2e-auq-consistency.test.ts': 1_080_000, - 'test/skill-e2e-auq-matrix.test.ts': 3_720_000, - 'test/skill-e2e-plan-format.test.ts': 2_600_000, + 'test/skill-e2e-auq-matrix.test.ts': 1_920_000, + 'test/skill-e2e-plan-format.test.ts': 1_360_000, 'test/skill-e2e-auto-decide-preserved.test.ts': 1_020_000, 'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_020_000, 'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_020_000, @@ -33,38 +33,22 @@ const expectedWalls = { 'test/skill-e2e-plan-mode-no-op.test.ts': 3_120_000, 'test/skill-e2e-plan-ceo-mode-routing.test.ts': 1_320_000, 'test/skill-e2e-plan-eng-plan-mode.test.ts': 1_320_000, - 'test/skill-e2e-plan-prosons.test.ts': 2_600_000, + 'test/skill-e2e-plan-prosons.test.ts': 1_360_000, 'test/skill-e2e-plan.test.ts': 3_720_000, }; -test('retry rule: only files whose every case is CAPTURE tier or shorter retry; longer cases run once', () => { - expect(RETRY_MAX_CASE_MS).toBe(ALL_TIERS.CAPTURE_MS + 15_000); +test('paid evals never retry: every paid file and registered row runs once', () => { for (const row of FILE_RETRY_BUDGETS) { - expect(row.retries, row.file).toBe(row.caseMs <= RETRY_MAX_CASE_MS ? (row.file.endsWith('plan-mode-no-op.test.ts') ? 2 : 1) : 0); - expect(retriesForFiles([row.file])).toBe(row.retries); + expect(Object.hasOwn(row, 'retries'), row.file).toBe(false); + expect(retriesForFiles([row.file])).toBe(0); } - expect(FILE_RETRY_BUDGETS.filter(row => row.retries > 0).map(row => row.file).sort()).toEqual([ - 'test/skill-e2e-auq-matrix.test.ts', 'test/skill-e2e-plan-format.test.ts', 'test/skill-e2e-plan-prosons.test.ts', - 'test/skill-e2e-qa-callers.test.ts', 'test/skill-llm-eval.test.ts', - ]); - const paid = collectPaidTestFiles(); - for (const file of SHORT_CASE_RETRY_FILES) { - expect(paid, `stale SHORT_CASE_RETRY_FILES entry: ${file}`).toContain(file); - expect(FILE_RETRY_BUDGETS.some(row => row.file === file)).toBe(false); - const source = read(file); - // Declared short budgets only: a JUDGE/CAPTURE tier or a literal at most the - // cap, no longer tier and no ms literal past the cap. - const literals = [...source.matchAll(/(? Number(match[1]!.replace(/_/g, ''))); - expect(/\b(?:JUDGE_MS|CAPTURE_MS)\b/.test(source) || literals.some(ms => ms >= 60_000 && ms <= RETRY_MAX_CASE_MS), file).toBe(true); - expect(source, file).not.toMatch(/\b(?:CAPTURE_LONG_MS|PTY_MS|PTY_LONG_MS|OVERLAY_CASE_[A-Z_]+)\b/); - expect(literals.filter(ms => ms > RETRY_MAX_CASE_MS && ms < 10_000_000), file).toEqual([]); - expect(retriesForFiles([file])).toBe(1); + for (const file of collectPaidTestFiles()) expect(retriesForFiles([file]), file).toBe(0); + expect(buildPaidShardArgs(['test/x.test.ts'], 1000, 2)).toContain('--retry'); + expect(buildPaidShardArgs(['test/x.test.ts'], 1000, 2).join(' ')).toContain('--retry 0'); + const scripts: Record = JSON.parse(read('package.json')).scripts; + for (const [name, command] of Object.entries(scripts)) { + if (/^test:(?:evals|e2e|gate|periodic)/.test(name)) expect(command, name).not.toMatch(/--retry(?:\s+|=)[1-9]/); } - for (const file of paid.filter(file => !SHORT_CASE_RETRY_FILES.includes(file) && !FILE_RETRY_BUDGETS.some(row => row.file === file))) { - expect(retriesForFiles([file]), file).toBe(0); - } - expect(retriesForFiles([SHORT_CASE_RETRY_FILES[0]!, 'test/skill-e2e-plan.test.ts'])).toBe(0); }); test('registration covers exactly the seventeen demonstrated full-file retry gaps', () => { @@ -133,14 +117,14 @@ for (const row of newBudgets) { outcomes: [{ files: [key], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count, skippedTests: 0, budget: resolvePaidShardBudget([key]) }] }]; - test(`${row.file}: full wall and existing retries propagate through planning`, () => { - expect(retriesForFiles([row.file])).toBe(row.retries); + test(`${row.file}: full wall propagates through planning and runs once`, () => { + expect(retriesForFiles([row.file])).toBe(0); expect(resolvePaidShardBudget([row.file])).toEqual({ timeoutMs: expectedWalls[row.file as keyof typeof expectedWalls], source: 'registered', policyId: row.id }); expect(planPaidShards(['test/a.test.ts', row.file, 'test/z.test.ts'], { maxFilesPerShard: 3 })).toContainEqual([row.file]); expect(() => resolvePaidShardBudget([row.file, 'test/neighbor.test.ts'])).toThrow('own shard'); expect(resolvePaidShardBudget([row.file], 50)).toEqual({ timeoutMs: 50, source: 'explicit', policyId: row.id }); expect(buildPaidShardArgs([row.file], row.shardMs, 2, retriesForFiles([row.file]))).toEqual([ - 'test', row.file, '--retry', String(row.retries), '--concurrent', '--max-concurrency=2', `--timeout=${row.shardMs}`, + 'test', row.file, '--retry', '0', '--concurrent', '--max-concurrency=2', `--timeout=${row.shardMs}`, ]); }); @@ -191,17 +175,17 @@ test('quality judge supervision includes the added judge without changing ordina expect(ALL_TIERS).toEqual({ JUDGE_MS: 120000, CAPTURE_MS: 300000, CAPTURE_LONG_MS: 600000, PTY_MS: 900000, PTY_LONG_MS: 1200000 }); const quality = 'test/skill-llm-eval.test.ts'; const qualityBudget = FILE_RETRY_BUDGETS.find(row => row.file === quality)!; - expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 6_220_000, source: 'registered', policyId: qualityBudget.id }); - expect(retriesForFiles([quality])).toBe(1); + expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 3_170_000, source: 'registered', policyId: qualityBudget.id }); + expect(retriesForFiles([quality])).toBe(0); const qualitySource = read(quality); const judgeTimeouts = [...qualitySource.matchAll(/}\s*,\s*(JUDGE_MS|WORKFLOW_JUDGE_TEST_MS)\s*\);/g)].map(match => match[1]); expect(judgeTimeouts.filter(timeout => timeout === 'JUDGE_MS')).toHaveLength(7); expect(judgeTimeouts.filter(timeout => timeout === 'WORKFLOW_JUDGE_TEST_MS')).toHaveLength(17); expect(qualitySource).toContain('WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000'); expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS'); - expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000); - expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([ - ...Array(2).fill([1, 1500000, 0, 1620000]), + expect(qualityBudget.shardMs).toBe(7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000) + 120_000); + expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.shardMs])).toEqual([ + ...Array(2).fill([1, 1500000, 1620000]), ]); for (const tier of ['gate', 'periodic'] as const) { const m = buildRunManifest({ tier, sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } }); @@ -227,7 +211,7 @@ test('detached PR fallback and release commands cover their actual default worke const prFloor = Math.ceil((Math.ceil(fullGateFiles.length / prWorkers) * 1_800_000 + fullGateFiles.reduce( (total, file) => total + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - 1_800_000), 0, )) / 1000 * 1.05); - expect(prFloor).toBe(77_501); + expect(prFloor).toBe(72_755); expect(prWall).toBe(92_820_000); expect(prWall).toBeGreaterThanOrEqual(paidShardWallUpperBoundMs(files, prWorkers) + 120_000); @@ -246,8 +230,8 @@ test('detached PR fallback and release commands cover their actual default worke )) / 1000 * 1.05)); } const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000; - expect(releaseFloors).toEqual([26_471, 30_797]); - expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(57_268); + expect(releaseFloors).toEqual([21_725, 22_481]); + expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(44_206); expect(detachedReleaseWall).toBe(116_700_000); expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000); }); @@ -301,7 +285,7 @@ test('both gate executors plan the complete census and supervise every planned s } }); -test('the periodic executor supervises every actual case and retry within its planned CI wall', () => { +test('the periodic executor supervises every actual case within its planned CI wall', () => { const workflow: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml')); const executor = workflow.jobs['eval-slices']; const emit = workflow.jobs['plan-slices'].steps.filter((step: any) => @@ -318,7 +302,7 @@ test('the periodic executor supervises every actual case and retry within its pl evalsAll: true, env: { EVALS_ALL: '1' } }); const census = manifest.entries.filter(row => row.status === 'planned'); expect(new Set(census.map(row => shardFile(row.file)))).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected)); - expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_220_000); + expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(3_170_000); const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder( census.filter(row => row.slice === i + 1)).map(row => row.file), active.jobs)); expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000); diff --git a/test/paid-run-manifest.test.ts b/test/paid-run-manifest.test.ts index 9539df224..b91da2547 100644 --- a/test/paid-run-manifest.test.ts +++ b/test/paid-run-manifest.test.ts @@ -487,8 +487,8 @@ describe('hollow-shard guard', () => { }); describe('retry parity', () => { - test('registered native workflows follow the retry rule while overlay attempts stay isolated', () => { - // A 25-minute case is past RETRY_MAX_CASE_MS: its timed-out attempt is the verdict. + test('registered native workflows and overlays run once', () => { + // Paid evals never retry: a timed-out attempt is the verdict. const native = 'test/skill-e2e-plan-ceo-split-overflow.test.ts'; expect(retriesForFiles([native])).toBe(0); expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(0); @@ -496,16 +496,15 @@ describe('retry parity', () => { const overlay = 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts'; expect(retriesForFiles([overlay])).toBe(0); }); - test('the matrix-era earned retries now follow the timeout-is-a-verdict rule, and each names a real file', () => { - // These three old matrix rows earned `retries: 2`; every one has a - // CAPTURE_LONG case, so a timed-out attempt is now their verdict. + test('the matrix-era earned retries are retired, and each names a real file', () => { + // These three old matrix rows earned `retries: 2`; paid evals never retry. for (const file of ['test/skill-e2e-office-hours-auto-mode.test.ts', 'test/skill-e2e-plan-mode-no-op.test.ts', 'test/skill-e2e-workflow.test.ts']) { expect(fs.existsSync(path.join(ROOT, file)), `stale retry parity entry: ${file}`).toBe(true); expect(retriesForFiles([file])).toBe(0); } expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(0); - expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1); + expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(0); expect(buildPaidShardArgs(['x'], 1000, 4, 2)).toContain('2'); - expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 1'); + expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 0'); }); }); diff --git a/test/review-finalization-budget.test.ts b/test/review-finalization-budget.test.ts index 84ed07275..d1eb0f5f4 100644 --- a/test/review-finalization-budget.test.ts +++ b/test/review-finalization-budget.test.ts @@ -1,4 +1,4 @@ -/** The real review registrations must finish capture cleanup before Bun retries. */ +/** The real review registrations record late results and clean up before finalization, under the production zero-retry arguments. */ import { expect, test } from 'bun:test'; import * as fs from 'node:fs'; import * as os from 'node:os'; @@ -12,7 +12,7 @@ const CASES = [ ['review-design-lite', 400, 35], ] as const; for (const [id, workMs, maxTurns] of CASES) { - test.each(['recover', 'both-timeout'])(`${id} records late results before retry or finalization: %s`, scenario => { + test.each(['success', 'timeout'])(`${id} records late results before finalization: %s`, scenario => { const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'review-finalization-')); const script = path.join(dir, 'registration.test.ts'); const facts = path.join(dir, 'events.jsonl'); @@ -41,7 +41,7 @@ mock.module(path.join(root, 'test/helpers/session-runner.ts'), () => ({ runSkillTest: async opts => { const id = ++attempts; event({ kind: 'start', id, timeout: opts.timeout, maxTurns: opts.maxTurns, cwd: opts.workingDirectory }); - const timeout = id === 1 || ${JSON.stringify(scenario)} === 'both-timeout'; + const timeout = ${JSON.stringify(scenario)} === 'timeout'; // Use the caller's actual work budget; only the provider and budget // constants are scaled. The actual registered Bun outer deadline stays. await new Promise(resolve => setTimeout(resolve, timeout ? opts.timeout + 50 : 80)); @@ -62,27 +62,25 @@ await import(path.join(root, ${JSON.stringify(PAID_FILE)})); `); try { const retries = retriesForFiles([PAID_FILE]); - expect(retries).toBe(1); + expect(retries).toBe(0); const child = Bun.spawnSync([process.execPath, ...buildPaidShardArgs([script], resolvePaidShardTimeoutMs([PAID_FILE]), 2, retries)], { cwd: ROOT, timeout: 15_000, stdout: 'pipe', stderr: 'pipe', env: { ...process.env, EVALS: '', EVALS_ALL: '', TMPDIR: dir, TMP: dir, TEMP: dir }, }); const output = child.stdout.toString() + child.stderr.toString(); - expect(child.exitCode, output).toBe(scenario === 'recover' ? 0 : 1); + expect(child.exitCode, output).toBe(scenario === 'success' ? 0 : 1); expect(output).not.toContain('Unhandled error between tests'); const events = fs.readFileSync(facts, 'utf8').trim().split('\n').map(line => JSON.parse(line)); const starts = events.filter(event => event.kind === 'start'); const ready = events.filter(event => event.kind === 'ready'); const records = events.filter(event => event.kind === 'record'); - expect(starts.map(event => event.id)).toEqual([1, 2]); + expect(starts.map(event => event.id)).toEqual([1]); expect(starts.map(({ timeout, maxTurns }) => ({ timeout, maxTurns }))) - .toEqual([{ timeout: workMs, maxTurns }, { timeout: workMs, maxTurns }]); - expect(ready).toEqual([{ kind: 'ready', id: 1, fixtureExists: true }, { kind: 'ready', id: 2, fixtureExists: true }]); + .toEqual([{ timeout: workMs, maxTurns }]); + expect(ready).toEqual([{ kind: 'ready', id: 1, fixtureExists: true }]); expect(records.map(event => [event.id, event.exitReason])) - .toEqual([[1, 'timeout'], [2, scenario === 'recover' ? 'success' : 'timeout']]); + .toEqual([[1, scenario === 'success' ? 'success' : 'timeout']]); expect(events.findIndex(event => event.kind === 'record' && event.id === 1)) - .toBeLessThan(events.findIndex(event => event.kind === 'start' && event.id === 2)); - expect(events.findIndex(event => event.kind === 'record' && event.id === 2)) .toBeLessThan(events.findIndex(event => event.kind === 'finalized')); expect(events.filter(event => event.kind === 'finalized')).toHaveLength(1); expect(events.find(event => event.kind === 'registration')).toEqual({ kind: 'registration', name: id, outerMs: workMs + 50 + 5_000 }); diff --git a/test/ship-hook-actor.test.ts b/test/ship-hook-actor.test.ts index 1f9627de8..f3c307cc5 100644 --- a/test/ship-hook-actor.test.ts +++ b/test/ship-hook-actor.test.ts @@ -13,9 +13,9 @@ import { DEFAULT_SHARD_TIMEOUT_MS, retriesForFiles } from '../scripts/test-paid- const cases: ShipHookCase[] = ['ship-managed-hook-refresh', 'ship-unmanaged-hook-consent', 'ship-local-hook-preservation']; type Fault = 'skip-guard' | 'skip-consent' | 'ask-overwrite' | 'direct-install' | 'read-receipts' | 'edit-policy' | 'tamper-receipts' | 'repeat-question' | 'rate-limit'; -test('whole-file supervision covers every F5 case and the unchanged Bun retry', () => { +test('whole-file supervision covers every F5 case run once', () => { for (const [file, count] of [['test/skill-e2e-ship-hook-refresh.test.ts', 1], ['test/skill-e2e-ship-hook-consent.test.ts', 2]] as const) { - expect(retriesForFiles([file])).toBe(1); + expect(retriesForFiles([file])).toBe(0); expect(count * CAPTURE_MS * (retriesForFiles([file]) + 1) + 120000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS); } }); diff --git a/test/ship-skip-actor.test.ts b/test/ship-skip-actor.test.ts index 605522fc3..a00c72095 100644 --- a/test/ship-skip-actor.test.ts +++ b/test/ship-skip-actor.test.ts @@ -152,9 +152,9 @@ function protocol(fault?: Fault, billing?: Array, controls: return { provider, directory: () => directory, calls: () => calls, sessions }; } -test('one bounded native case preserves the existing whole-file retry allowance', () => { +test('one bounded native case fits the whole-file wall and never retries', () => { const retries = retriesForFiles(['test/skill-e2e-ship-skip.test.ts']); - expect(retries).toBe(1); + expect(retries).toBe(0); expect(CAPTURE_MS * (retries + 1) + 120000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS); }); diff --git a/test/workflow-boundaries-fixture.test.ts b/test/workflow-boundaries-fixture.test.ts index f94961a3a..0db8cfb06 100644 --- a/test/workflow-boundaries-fixture.test.ts +++ b/test/workflow-boundaries-fixture.test.ts @@ -292,12 +292,12 @@ test('F9 changed-input selection produces three cases with exact patterns and co expect(prProfileTestNamePattern(files[1], selected.selection)).toBe('(?:^|\\s)(?:investigate-owned-abort|investigate-owned-ending-error)$'); }); -test('both F9 files fit the existing wall with every Bun retry and reserve', () => { +test('both F9 files fit the existing wall with their one run and reserve', () => { for (const file of files) { const source = fs.readFileSync(path.join(import.meta.dir, '..', file), 'utf8'); const count = PR_PROFILE_FILES[file].length; expect([...source.matchAll(/\}, CAPTURE_MS\);/g)]).toHaveLength(count); - expect(retriesForFiles([file])).toBe(1); + expect(retriesForFiles([file])).toBe(0); const budget = resolvePaidShardBudget([file]); expect(budget).toEqual({ timeoutMs: DEFAULT_SHARD_TIMEOUT_MS, source: 'default', policyId: null }); expect(count * CAPTURE_MS * (retriesForFiles([file]) + 1) + 120000).toBeLessThanOrEqual(budget.timeoutMs);