diff --git a/.github/workflows/evals-marathon.yml b/.github/workflows/evals-marathon.yml new file mode 100644 index 000000000..4a7d3a0df --- /dev/null +++ b/.github/workflows/evals-marathon.yml @@ -0,0 +1,298 @@ +name: Marathon Evals +# The NON-BLOCKING marathon lane: complete start-to-finish flows (tier +# 'marathon' in test/helpers/touchfiles-data.ts / describeE2ETier('marathon')) +# that take longer than a blocking lane's ~12-minute wall. They never run in +# the PR gate (evals.yml) or the weekly periodic + gate census +# (evals-periodic.yml); nothing requires this workflow, so a red marathon +# reports through its own tracking issue without gating any merge. Same engine +# and FAIL-CLOSED report as the other lanes: one planner manifest, one file per +# runner, a missing slice artifact is a failure. Always fresh: no result reuse. +on: + schedule: + - cron: '0 12 * * 6' # Saturday 12:00 UTC, clear of the Monday periodic census + workflow_dispatch: + +concurrency: + group: evals-marathon + cancel-in-progress: true + +env: + IMAGE: ghcr.io/${{ github.repository }}/ci + EVALS_PROFILE: full + EVALS_FRESH: "1" + EVALS_CACHE_PURPOSE: marathon + +jobs: + IMAGE: ghcr.io/${{ github.repository }}/ci + EVALS_PROFILE: full + EVALS_FRESH: "1" + EVALS_CACHE_PURPOSE: periodic + +jobs: + build-image: + runs-on: ubicloud-standard-8 + timeout-minutes: 15 + permissions: + contents: read + packages: write + outputs: + image-tag: ${{ steps.meta.outputs.tag }} + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + + - id: meta + # Keep in sync with evals.yml and evals-periodic.yml — key on Dockerfile + lockfile only + # (package.json's version field would bust the key on every ship). + # Byte-identity pinned by test/ci-image-tag-binding.test.ts. + run: echo "tag=${{ env.IMAGE }}:${{ hashFiles('.github/docker/Dockerfile.ci', 'bun.lock', 'patches/**') }}" >> "$GITHUB_OUTPUT" + + - uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4 + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + + - name: Check if image exists + id: check + run: | + if docker manifest inspect ${{ steps.meta.outputs.tag }} > /dev/null 2>&1; then + echo "exists=true" >> "$GITHUB_OUTPUT" + else + echo "exists=false" >> "$GITHUB_OUTPUT" + fi + + - if: steps.check.outputs.exists == 'false' + run: cp package.json bun.lock .github/docker/ && cp -R patches .github/docker/patches + + # Registry cache export needs a docker-container builder — the default + # `docker` driver hard-errors on cache-to. + - if: steps.check.outputs.exists == 'false' + uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4 + + - if: steps.check.outputs.exists == 'false' + uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7 + with: + context: .github/docker + file: .github/docker/Dockerfile.ci + push: true + # Cron-triggered in the base repo only, so cache export is always safe here. + cache-from: type=registry,ref=${{ env.IMAGE }}:buildcache + cache-to: type=registry,ref=${{ env.IMAGE }}:buildcache,mode=max + tags: | + ${{ steps.meta.outputs.tag }} + ${{ env.IMAGE }}:latest + + + plan-slices: + runs-on: ubicloud-standard-8 + timeout-minutes: 10 + permissions: + contents: read + outputs: + slices: ${{ steps.matrix.outputs.slices }} + timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }} + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + + - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 + with: + bun-version: 1.4.0 + + # One marathon file per runner: a 1-second budget never packs two + # recorded files together. + - name: Emit run manifest (ALL marathon tests) + env: + EVALS_ALL: "1" + run: EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --emit-plan /tmp/marathon-plan/manifest.json --slice-budget 1 --jobs 1 + + - name: Derive the executor matrix from the plan + id: matrix + run: | + echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT" + echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT" + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: marathon-plan + path: /tmp/marathon-plan/manifest.json + retention-days: 30 + + eval-slices: + runs-on: ubicloud-standard-8 + needs: [build-image, plan-slices] + env: + EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-marathon-${{ matrix.slice }} + # One marathon file per runner; the job timeout is the plan's supervised + # worst case plus 20 minutes setup/upload. + timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }} + permissions: + contents: read + packages: read + container: + image: ${{ needs.build-image.outputs.image-tag }} + credentials: + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + options: --user runner + strategy: + fail-fast: false + max-parallel: 8 + matrix: + slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }} + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + # Full history: files with SELF-derived selection (the LLM-judge + # map, routing) walk git at module load, and selection is + # fail-closed on git errors — a shallow checkout crashed those + # shards on the lane's first live run ("ambiguous argument + # 'main...HEAD'"). The manifest still governs WHICH shards run. + fetch-depth: 0 + persist-credentials: false + + - name: Fix bun temp + uses: ./.github/actions/fix-bun-temp + + - name: Restore deps + uses: ./.github/actions/restore-deps + + - run: bun run build + + # Any slice can host a PTY test — seed + registration run + # unconditionally (idempotent; mirrors evals.yml's sliced lane). The + # register composite carries the fail-fast dangling-symlink/frontmatter + # verification loop — this lane previously LACKED it, so a moved skill + # target surfaced as a silent "Unknown command" + wedged PTY session. + - name: Seed claude interactive config + uses: ./.github/actions/seed-claude-config + with: + anthropic-api-key: ${{ secrets.ANTHROPIC_API_KEY }} + + - name: Register gstack skills for PTY tests + uses: ./.github/actions/register-gstack-skills + + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 + with: + name: marathon-plan + path: /tmp/marathon-plan + + - name: Run marathon slice ${{ matrix.slice }} + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} + GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }} + PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers + EVALS_JOBS: "1" + EVALS_CONCURRENCY: "2" + GSTACK_EVAL_DIR: /tmp/marathon-slice-results + run: EVALS_TIER=marathon bun run scripts/test-paid-shards.ts --tier marathon --plan /tmp/marathon-plan/manifest.json --slice ${{ matrix.slice }} + + - name: Upload slice results + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: marathon-slice-${{ matrix.slice }} + path: /tmp/marathon-slice-results + retention-days: 90 + + - name: Upload native capture evidence + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: native-captures-${{ env.EVALS_RUN_ID }} + include-hidden-files: true + path: | + ~/.gstack/projects/*/e2e-runs + ~/.gstack/projects/*/evals/qa-callers + ~/.gstack-dev/e2e-runs + ~/.gstack-dev/evals/qa-callers + if-no-files-found: ignore + retention-days: 90 + + - name: Upload shard logs on failure + if: failure() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: marathon-slice-${{ matrix.slice }}-logs + include-hidden-files: true + # The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the + # runner's spool lands THERE, not /tmp — the original /tmp glob + # uploaded nothing and a red slice's diagnostics were unreachable. + path: | + /home/runner/.cache/gstack-paid-shard-*.log + /tmp/gstack-paid-shard-*.log + if-no-files-found: ignore + retention-days: 30 + + report: + runs-on: ubicloud-standard-2 + needs: [plan-slices, eval-slices] + # !cancelled(): the report must run (and FAIL) when an executor died — a + # missing slice artifact reading as green is the class this lane kills — + # but a cancelled run stops here. + if: ${{ !cancelled() && needs.plan-slices.result == 'success' }} + timeout-minutes: 10 + permissions: + contents: read + issues: write + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + + - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 + with: + bun-version: 1.4.0 + + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 + with: + name: marathon-plan + path: /tmp/marathon-report + + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 + with: + pattern: marathon-slice-[0-9]* + path: /tmp/marathon-report + merge-multiple: true + + - name: Reconcile slices against the manifest (fail-closed) + id: reconcile + if: always() + run: | + set +e + EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --report /tmp/marathon-report | tee /tmp/report.txt + # PIPESTATUS[0], NOT $?: the default step shell has no pipefail. + echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT" + + # One tracking issue for the whole lane (never one per week). + - name: Upsert tracking issue on failure + if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success') + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + set -euo pipefail + TITLE="Weekly marathon evals: red lane needs triage" + BODY_FILE=/tmp/issue-body.md + { + echo "Automated weekly marathon report (non-blocking lane) — run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}" + echo + echo "- reconciliation exit: ${{ steps.reconcile.outputs.exit }}" + echo "- marathon slices job: ${{ needs.eval-slices.result }}" + echo + echo '```' + tail -c 6000 /tmp/report.txt 2>/dev/null || echo "(no reconciliation output)" + echo '```' + } > "$BODY_FILE" + EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty') + if [ -n "$EXISTING" ]; then + gh issue comment "$EXISTING" --repo "$GITHUB_REPOSITORY" --body-file "$BODY_FILE" + echo "commented on #$EXISTING" + else + gh issue create --repo "$GITHUB_REPOSITORY" --title "$TITLE" --body-file "$BODY_FILE" + fi + + - name: Fail the workflow when reconciliation failed + if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success') + run: exit 1 diff --git a/.github/workflows/evals-periodic.yml b/.github/workflows/evals-periodic.yml index 8272c4abb..a684b7b5b 100644 --- a/.github/workflows/evals-periodic.yml +++ b/.github/workflows/evals-periodic.yml @@ -4,8 +4,13 @@ name: Periodic Evals # tests can't rot invisibly — the class where the autoplan-dual-voice E2E was # silently broken for months until a lucky local diff selected it. Engine: # scripts/test-paid-shards.ts (the same runner local eval:bg:periodic uses): -# one planner manifest, 6 ordinary slices plus an overlay slice, and a FAIL-CLOSED report — a slice -# whose artifact never landed is a failure, not an absence. The gate-census +# one planner manifest packed by recorded durations into as many ~9-minute +# executors as the work needs (one file, or a tightly packed group, per +# runner; overlays share one final slice), and a FAIL-CLOSED report — a slice +# whose artifact never landed is a failure, not an absence. The matrix size +# and job timeout come from the plan, so they cannot drift from the census. +# Full end-to-end flows run in the non-blocking marathon lane +# (evals-marathon.yml), never here. The gate-census # job is the weekly EVALS_ALL backstop for the gate tier (PR lanes are # diff-billed, so without it the full gate census might never execute # anywhere); the hollow-shard guard (exit 0 + zero executed tests under @@ -84,6 +89,11 @@ jobs: timeout-minutes: 10 permissions: contents: read + outputs: + periodic_slices: ${{ steps.periodic-matrix.outputs.slices }} + periodic_timeout_minutes: ${{ steps.periodic-matrix.outputs.timeout_minutes }} + gate_slices: ${{ steps.gate-matrix.outputs.slices }} + gate_timeout_minutes: ${{ steps.gate-matrix.outputs.timeout_minutes }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: @@ -96,7 +106,13 @@ jobs: - name: Emit run manifest (ALL periodic tests minus reasoned excludes) env: EVALS_ALL: "1" - run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 7 + run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 + + - name: Derive the periodic executor matrix from the plan + id: periodic-matrix + run: | + echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT" + echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT" - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: @@ -107,7 +123,13 @@ jobs: - name: Emit gate census manifest (ALL gate tests) env: EVALS_ALL: "1" - run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 7 --skip-judges + run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges + + - name: Derive the gate census executor matrix from the plan + id: gate-matrix + run: | + echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/gate-census-plan/manifest.json)" >> "$GITHUB_OUTPUT" + echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/gate-census-plan/manifest.json)" >> "$GITHUB_OUTPUT" - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: @@ -120,9 +142,10 @@ jobs: needs: [build-image, plan-slices] env: EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }} - # Seven slices retain every registered case and retry. The complete - # census needs at most 244m40s per slice, plus 20 minutes setup/upload. - timeout-minutes: 360 + # The planner packs ~9 minutes of recorded work per slice; the job timeout + # is its supervised worst case (every shard at its wall) plus 20 minutes + # setup/upload, computed from the same manifest the slices execute. + timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }} permissions: contents: read packages: read @@ -134,9 +157,11 @@ jobs: options: --user runner strategy: fail-fast: false - max-parallel: 8 + # Every planned slice starts at once; test/evals-workflow-wiring.test.ts + # fails when the live plan outgrows this cap. + max-parallel: 24 matrix: - slice: [1, 2, 3, 4, 5, 6, 7] + slice: ${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: @@ -174,7 +199,7 @@ jobs: name: paid-plan path: /tmp/paid-plan - - name: Run slice ${{ matrix.slice }}/7 + - name: Run periodic slice ${{ matrix.slice }} env: ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} @@ -231,8 +256,8 @@ jobs: needs: [build-image, plan-slices] env: EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-gate-census-${{ matrix.slice }} - # Seven slices need at most 272m each, plus 20 minutes setup/upload. - timeout-minutes: 352 + # Supervised worst case of the packed plan plus 20 minutes setup/upload. + timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.gate_timeout_minutes) }} permissions: contents: read packages: read @@ -243,11 +268,11 @@ jobs: password: ${{ secrets.GITHUB_TOKEN }} options: --user runner strategy: - # Four file workers total, each retaining two in-file case workers. + # Two file workers per slice, each retaining two in-file case workers. fail-fast: false - max-parallel: 4 + max-parallel: 16 matrix: - slice: [1, 2, 3, 4, 5, 6, 7] + slice: ${{ fromJSON(needs.plan-slices.outputs.gate_slices) }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: @@ -272,13 +297,13 @@ jobs: name: gate-census-plan path: /tmp/gate-census-plan - - name: Run gate census slice ${{ matrix.slice }}/7 + - name: Run gate census slice ${{ matrix.slice }} env: ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }} PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers - EVALS_JOBS: "1" + EVALS_JOBS: "2" EVALS_CONCURRENCY: "2" GSTACK_EVAL_DIR: /tmp/gate-census-results run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/gate-census-plan/manifest.json --slice ${{ matrix.slice }} diff --git a/.github/workflows/evals.yml b/.github/workflows/evals.yml index b0d97ad2c..87775320b 100644 --- a/.github/workflows/evals.yml +++ b/.github/workflows/evals.yml @@ -125,6 +125,9 @@ jobs: timeout-minutes: 10 permissions: contents: read + outputs: + slices: ${{ steps.matrix.outputs.slices }} + timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: @@ -141,7 +144,7 @@ jobs: if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all' env: EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }} - run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slices 7 + run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 - name: Emit validation-phase manifest if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all' @@ -152,23 +155,30 @@ jobs: run: | bun --no-install -e ' import { mkdirSync, writeFileSync } from "node:fs"; - import { buildRunManifest, collectPaidTestFiles } from "./scripts/test-paid-shards.ts"; + import { buildRunManifest, collectPaidTestFiles, restrictManifestSelection } from "./scripts/test-paid-shards.ts"; const phase = process.env.VALIDATION_PHASE; if (!["quality", "cookie-quality", "behavior", "cookie-behavior"].includes(phase)) throw new Error("Invalid validation phase"); const cookieBehavior = phase === "cookie-behavior"; const discovered = phase === "cookie-quality" ? ["test/skill-llm-eval.test.ts"] : cookieBehavior ? ["test/skill-e2e-bws.test.ts", "test/skill-e2e-qa-workflow.test.ts", "test/skill-e2e-design.test.ts", "test/skill-e2e-diagram.test.ts", "test/skill-e2e-deploy.test.ts"] : collectPaidTestFiles().filter(file => file.startsWith("test/skill-llm-eval") === (phase === "quality")); - const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceCount: 6, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered, + const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceBudgetMs: 540000, jobs: 2, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered, ...(cookieBehavior ? { changedFiles: ["browse/src/cookie-picker-routes.ts", "browse/src/cookie-import-browser.ts", "browse/src/bun-polyfill.cjs"], env: { ...process.env, EVALS_ALL: "" } } : {}) }); - if (phase === "cookie-quality") manifest.selection = { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] }; - if (cookieBehavior) manifest.selection = { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] }; - manifest.selectionReason = phase + " validation subset; " + manifest.selectionReason; + const subset = phase === "cookie-quality" ? { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] } + : cookieBehavior ? { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] } : null; + const restricted = subset ? restrictManifestSelection(manifest, subset, "outside the " + phase + " validation subset") : manifest; + restricted.selectionReason = phase + " validation subset; " + manifest.selectionReason; mkdirSync("/tmp/paid-plan", { recursive: true }); - writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(manifest, null, 2) + "\n"); - console.log(phase + ": " + manifest.entries.filter(entry => entry.status === "planned").length + " planned shards"); + writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(restricted, null, 2) + "\n"); + console.log(phase + ": " + restricted.entries.filter(entry => entry.status === "planned").length + " planned shards"); ' + - name: Derive the executor matrix from the plan + id: matrix + run: | + echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT" + echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT" + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: paid-plan @@ -184,14 +194,11 @@ jobs: # (image already published), but a newer push's cancel-in-progress stops # it instead of letting a superseded run finish its paid slices first. if: ${{ !cancelled() && needs.build-image.result == 'success' && needs.plan-slices.result == 'success' }} - # Aggregate spawn-concurrency budget: 6 slices x EVALS_JOBS=2 x - # EVALS_CONCURRENCY=2 = 24 concurrent tests lane-wide (the old matrix's - # 40-way per row queued claude session STARTUP behind 39 siblings and ate - # per-test budgets — the documented timeout-flake family). Tune with - # parity data before raising. - # The complete gate census needs at most 242 minutes per slice; keep - # 20 minutes for setup/upload without preempting configured retries. - timeout-minutes: 265 + # The planner packs ~9 minutes of recorded work per slice (EVALS_JOBS=2 x + # EVALS_CONCURRENCY=2 per runner, never the old 40-way per-row fan-out + # that queued claude session STARTUP behind 39 siblings). The job timeout + # is the plan's supervised worst case plus 20 minutes setup/upload. + timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }} permissions: contents: read packages: read @@ -203,9 +210,9 @@ jobs: options: --user runner strategy: fail-fast: false - max-parallel: 6 + max-parallel: 16 matrix: - slice: [1, 2, 3, 4, 5, 6, 7] + slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: @@ -246,7 +253,7 @@ jobs: # Only this PR's receipts are eligible. No base-branch or cross-PR restore # prefix; every receipt also verifies exact inputs and its original age. - - name: Restore this PR's verified judge results + - name: Restore this PR's verified judge and E2E results if: github.event_name == 'pull_request' uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 with: @@ -254,7 +261,7 @@ jobs: key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }} restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}- - - name: Run slice ${{ matrix.slice }}/7 + - name: Run slice ${{ matrix.slice }} env: ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} @@ -285,7 +292,7 @@ jobs: # An unrelated failing case does not discard already verified passes. # Failed/retried/partial attempts never become receipts in the first place. - - name: Save verified judge results for this PR + - name: Save verified judge and E2E results for this PR if: ${{ !cancelled() && steps.receipts.outputs.present == 'true' }} uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 with: @@ -531,7 +538,7 @@ jobs: --- - *Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → 6 executors → fail-closed report). Reused scores retain their original provenance and expiry.*" + *Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → duration-packed executors → fail-closed report). Reused results retain their original provenance and expiry.*" if [ "$FAILED" -gt 0 ]; then FAILURES="" diff --git a/scripts/paid-test-durations.json b/scripts/paid-test-durations.json index 47cbdedd6..e175368e7 100644 --- a/scripts/paid-test-durations.json +++ b/scripts/paid-test-durations.json @@ -1,44 +1,178 @@ { - "version": 1, - "recordedAt": "2026-09-28T23:00:00Z", - "durations": { - "test/skill-e2e-ask-user-question-format-compliance.test.ts": 67000, - "test/skill-e2e-bws.test.ts": 99000, - "test/skill-e2e-coverage-audit.test.ts": 60000, - "test/skill-e2e-cso.test.ts": 220000, - "test/skill-e2e-deploy.test.ts": 496000, - "test/skill-e2e-design.test.ts": 228000, - "test/skill-e2e-diagram.test.ts": 25000, - "test/skill-e2e-docsync-spawned.test.ts": 49000, - "test/skill-e2e-hermetic-canary.test.ts": 5000, - "test/skill-e2e-investigate-owned-completion.test.ts": 50000, - "test/skill-e2e-investigate-owned-termination.test.ts": 49000, - "test/skill-e2e-learnings.test.ts": 32000, - "test/skill-e2e-office-hours-auto-mode.test.ts": 61000, - "test/skill-e2e-plan-ceo-finding-floor.test.ts": 233000, - "test/skill-e2e-plan-ceo-plan-mode.test.ts": 35000, - "test/skill-e2e-plan-design-with-ui.test.ts": 435000, - "test/skill-e2e-plan-devex-finding-floor.test.ts": 187000, - "test/skill-e2e-plan-devex-plan-mode.test.ts": 103000, - "test/skill-e2e-plan-mode-no-op.test.ts": 206000, - "test/skill-e2e-plan-tune.test.ts": 58000, - "test/skill-e2e-plan.test.ts": 312000, - "test/skill-e2e-qa-workflow.test.ts": 437000, - "test/skill-e2e-retro.test.ts": 139000, - "test/skill-e2e-review-army.test.ts": 568000, - "test/skill-e2e-review-attribution.test.ts": 80000, - "test/skill-e2e-review.test.ts": 135000, - "test/skill-e2e-session-intelligence.test.ts": 47000, - "test/skill-e2e-shared-libs-paths.test.ts": 713000, - "test/skill-e2e-shared-libs.test.ts": 662000, - "test/skill-e2e-ship-docsync.test.ts": 148000, - "test/skill-e2e-ship-hook-consent.test.ts": 72000, - "test/skill-e2e-ship-hook-refresh.test.ts": 66000, - "test/skill-e2e-skillify.test.ts": 156000, - "test/skill-e2e-third-party-actions.test.ts": 75000, - "test/skill-e2e-triage.test.ts": 49000, - "test/skill-e2e-workflow.test.ts": 234000, - "test/skill-llm-eval.test.ts": 103000, - "test/skill-routing-e2e.test.ts": 1000 + "version": 2, + "recordedAt": "2026-09-28T09:05:00Z", + "source": "periodic census run 36385945043 (eval-slices = periodic, gate-census = gate); timed-out shards record their wall, self-skipping shards 1s; case shards (#) from the Bun per-case walls in that run (gate census log) and the eval JSON slowest cases (periodic)", + "tiers": { + "gate": { + "test/llm-judge-recommendation.test.ts": 1000, + "test/skill-e2e-ask-user-question-format-compliance.test.ts": 60000, + "test/skill-e2e-autoplan-dual-voice.test.ts": 1000, + "test/skill-e2e-bws.test.ts": 82000, + "test/skill-e2e-context-skills.test.ts": 1000, + "test/skill-e2e-coverage-audit.test.ts": 46000, + "test/skill-e2e-cso.test.ts": 251000, + "test/skill-e2e-deploy.test.ts": 427000, + "test/skill-e2e-design.test.ts": 261000, + "test/skill-e2e-design.test.ts#design-review-detector-shim": 51000, + "test/skill-e2e-design.test.ts#design-review-detector-shim-dom": 108000, + "test/skill-e2e-design.test.ts#design-review-plugin-handoff": 110000, + "test/skill-e2e-design.test.ts#plan-design-review-no-ui-scope": 43000, + "test/skill-e2e-diagram.test.ts": 32000, + "test/skill-e2e-docsync-spawned.test.ts": 51000, + "test/skill-e2e-first-task-scaffold.test.ts": 1000, + "test/skill-e2e-gbrain-roundtrip-local.test.ts": 1000, + "test/skill-e2e-hermetic-canary.test.ts": 8000, + "test/skill-e2e-investigate-owned-completion.test.ts": 45000, + "test/skill-e2e-investigate-owned-termination.test.ts": 44000, + "test/skill-e2e-ios-device.test.ts": 1000, + "test/skill-e2e-learnings.test.ts": 35000, + "test/skill-e2e-office-hours-auto-mode.test.ts": 67000, + "test/skill-e2e-office-hours-brain-writeback.test.ts": 1000, + "test/skill-e2e-office-hours-phase4.test.ts": 1000, + "test/skill-e2e-office-hours.test.ts": 1000, + "test/skill-e2e-plan-ceo-finding-floor.test.ts": 516000, + "test/skill-e2e-plan-ceo-plan-mode.test.ts": 39000, + "test/skill-e2e-plan-decision-classification.test.ts": 1000, + "test/skill-e2e-plan-design-with-ui.test.ts": 500000, + "test/skill-e2e-plan-devex-finding-floor.test.ts": 233000, + "test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 1000, + "test/skill-e2e-plan-devex-plan-mode.test.ts": 85000, + "test/skill-e2e-plan-format.test.ts": 1000, + "test/skill-e2e-plan-mode-no-op.test.ts": 206000, + "test/skill-e2e-plan-prosons.test.ts": 1000, + "test/skill-e2e-plan-tune.test.ts": 60000, + "test/skill-e2e-plan.test.ts": 251000, + "test/skill-e2e-plan.test.ts#codex-offered-ceo-review": 55000, + "test/skill-e2e-plan.test.ts#codex-offered-design-review": 59000, + "test/skill-e2e-plan.test.ts#codex-offered-eng-review": 58000, + "test/skill-e2e-plan.test.ts#codex-offered-office-hours": 54000, + "test/skill-e2e-plan.test.ts#office-hours-spec-review": 34000, + "test/skill-e2e-plan.test.ts#plan-ceo-review-benefits": 52000, + "test/skill-e2e-plan.test.ts#plan-review-report": 51000, + "test/skill-e2e-qa-bugs.test.ts": 1000, + "test/skill-e2e-qa-workflow.test.ts": 398000, + "test/skill-e2e-retro.test.ts": 152000, + "test/skill-e2e-review-army.test.ts": 520000, + "test/skill-e2e-review-army.test.ts#review-army-delivery-audit": 48000, + "test/skill-e2e-review-army.test.ts#review-army-json-findings": 24000, + "test/skill-e2e-review-army.test.ts#review-army-migration-safety": 140000, + "test/skill-e2e-review-army.test.ts#review-army-perf-n-plus-one": 244000, + "test/skill-e2e-review-army.test.ts#review-army-quality-score": 64000, + "test/skill-e2e-review-attribution.test.ts": 87000, + "test/skill-e2e-review.test.ts": 129000, + "test/skill-e2e-session-intelligence.test.ts": 59000, + "test/skill-e2e-shared-libs-paths.test.ts": 680000, + "test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-index-flags": 201000, + "test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-path-eligibility": 252000, + "test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-prior-coverage": 227000, + "test/skill-e2e-shared-libs.test.ts": 658000, + "test/skill-e2e-shared-libs.test.ts#shared-libs-read-only": 230000, + "test/skill-e2e-shared-libs.test.ts#shared-libs-review-lifecycle": 278000, + "test/skill-e2e-shared-libs.test.ts#shared-libs-review-revalidation": 427000, + "test/skill-e2e-shared-libs.test.ts#shared-libs-unsupported-git": 168000, + "test/skill-e2e-ship-docsync.test.ts": 129000, + "test/skill-e2e-ship-hook-consent.test.ts": 63000, + "test/skill-e2e-ship-hook-refresh.test.ts": 70000, + "test/skill-e2e-skillify.test.ts": 188000, + "test/skill-e2e-third-party-actions.test.ts": 70000, + "test/skill-e2e-triage.test.ts": 80000, + "test/skill-e2e-workflow.test.ts": 300000, + "test/skill-llm-eval.test.ts": 354000, + "test/skill-routing-e2e.test.ts": 1000 + }, + "periodic": { + "test/carve-section-loading-browse.test.ts": 109000, + "test/carve-section-loading-codex.test.ts": 118000, + "test/carve-section-loading-design-consultation.test.ts": 349000, + "test/carve-section-loading-design-html.test.ts": 259000, + "test/carve-section-loading-design-shotgun.test.ts": 194000, + "test/carve-section-loading-document-release.test.ts": 95000, + "test/carve-section-loading-land-and-deploy.test.ts": 185000, + "test/carve-section-loading-plan-design-review.test.ts": 284000, + "test/carve-section-loading-plan-devex-review.test.ts": 396000, + "test/carve-section-loading-plan-eng-review.test.ts": 301000, + "test/carve-section-loading-qa.test.ts": 114000, + "test/carve-section-loading-retro.test.ts": 230000, + "test/carve-section-loading-review.test.ts": 174000, + "test/carve-section-loading-setup-gbrain.test.ts": 91000, + "test/carve-section-loading-spec.test.ts": 150000, + "test/codex-e2e-recommendation-substance.test.ts": 1000, + "test/codex-e2e-shared-libs.test.ts": 1000, + "test/codex-e2e-sol-scope.test.ts": 1000, + "test/codex-e2e.test.ts": 1000, + "test/llm-judge-recommendation.test.ts": 15000, + "test/skill-e2e-arm-benchmark.test.ts": 76000, + "test/skill-e2e-aside.test.ts": 1000, + "test/skill-e2e-auq-consistency.test.ts": 75000, + "test/skill-e2e-auq-matrix.test.ts": 320000, + "test/skill-e2e-auq-verbose-vs-carved-ab.test.ts": 66000, + "test/skill-e2e-auto-decide-preserved.test.ts": 113000, + "test/skill-e2e-autoplan-dual-voice.test.ts": 532000, + "test/skill-e2e-benchmark-providers.test.ts": 9000, + "test/skill-e2e-bws.test.ts": 1000, + "test/skill-e2e-context-skills.test.ts": 181000, + "test/skill-e2e-coverage-audit.test.ts": 1000, + "test/skill-e2e-cso.test.ts": 358000, + "test/skill-e2e-deploy.test.ts": 1000, + "test/skill-e2e-design.test.ts": 817000, + "test/skill-e2e-design.test.ts#design-consultation-core": 204000, + "test/skill-e2e-design.test.ts#design-html-slop-gate": 205000, + "test/skill-e2e-design.test.ts#plan-design-review-plan-mode": 283000, + "test/skill-e2e-diagram.test.ts": 166000, + "test/skill-e2e-first-task-scaffold.test.ts": 9000, + "test/skill-e2e-gbrain-roundtrip-local.test.ts": 1000, + "test/skill-e2e-health.test.ts": 165000, + "test/skill-e2e-hermetic-canary.test.ts": 1000, + "test/skill-e2e-ios-device.test.ts": 1000, + "test/skill-e2e-learnings.test.ts": 1000, + "test/skill-e2e-office-hours-brain-writeback.test.ts": 174000, + "test/skill-e2e-office-hours-phase4.test.ts": 60000, + "test/skill-e2e-office-hours-section-loading.test.ts": 1200000, + "test/skill-e2e-office-hours.test.ts": 119000, + "test/skill-e2e-outside-plan-disabled.test.ts": 43000, + "test/skill-e2e-outside-voice.test.ts": 1000, + "test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts": 81000, + "test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts": 66000, + "test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts": 42000, + "test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts": 278000, + "test/skill-e2e-plan-ceo-mode-routing.test.ts": 575000, + "test/skill-e2e-plan-ceo-review-section-loading.test.ts": 604000, + "test/skill-e2e-plan-ceo-split-overflow.test.ts": 1332000, + "test/skill-e2e-plan-decision-classification.test.ts": 122000, + "test/skill-e2e-plan-design-finding-floor.test.ts": 175000, + "test/skill-e2e-plan-design-plan-mode.test.ts": 117000, + "test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 83000, + "test/skill-e2e-plan-eng-finding-floor.test.ts": 282000, + "test/skill-e2e-plan-eng-multi-finding-batching.test.ts": 734000, + "test/skill-e2e-plan-eng-plan-mode.test.ts": 91000, + "test/skill-e2e-plan-format.test.ts": 282000, + "test/skill-e2e-plan-prosons.test.ts": 180000, + "test/skill-e2e-plan-tune.test.ts": 1000, + "test/skill-e2e-plan.test.ts": 824000, + "test/skill-e2e-plan.test.ts#plan-ceo-review": 123000, + "test/skill-e2e-plan.test.ts#plan-ceo-review-selective": 244000, + "test/skill-e2e-plan.test.ts#plan-eng-review-artifact": 153000, + "test/skill-e2e-qa-bugs.test.ts": 283000, + "test/skill-e2e-qa-workflow.test.ts": 239000, + "test/skill-e2e-retro.test.ts": 114000, + "test/skill-e2e-review-army.test.ts": 511000, + "test/skill-e2e-review-army.test.ts#review-army-consensus": 278000, + "test/skill-e2e-review-army.test.ts#review-army-simplification": 144000, + "test/skill-e2e-review-attribution.test.ts": 1000, + "test/skill-e2e-review.test.ts": 131000, + "test/skill-e2e-session-intelligence.test.ts": 1000, + "test/skill-e2e-setup-gbrain-bad-token.test.ts": 49000, + "test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts": 119000, + "test/skill-e2e-setup-gbrain-remote.test.ts": 110000, + "test/skill-e2e-shared-libs-periodic.test.ts": 415000, + "test/skill-e2e-ship-section-loading.test.ts": 379000, + "test/skill-e2e-skillify.test.ts": 56000, + "test/skill-e2e-sync-gbrain-readiness.test.ts": 74000, + "test/skill-e2e-third-party-actions.test.ts": 1000, + "test/skill-e2e-triage.test.ts": 1000, + "test/skill-e2e-workflow.test.ts": 1000, + "test/skill-llm-eval.test.ts": 402000, + "test/skill-routing-e2e.test.ts": 57000 + } } } diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 38d75d19e..15ef2177a 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -67,12 +67,15 @@ import { } from './test-strict-output'; import { PAID_TEST_GLOBS, isPaidTestFile } from '../test/helpers/paid-test-set'; import { PERIODIC_CI_EXCLUDE } from '../test/helpers/periodic-exclude-data'; -import { FILE_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets'; +import { FILE_RETRY_BUDGETS, SHORT_CASE_RETRY_FILES, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets'; import { getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile, evalEntryOutcome } from '../test/helpers/eval-store'; import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-review'; import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight'; import { OVERLAY_MIN_FILE_WALL_MS } from '../test/helpers/overlay-case-policy'; import { PR_PROFILE_CASE_IDS, PR_PROFILE_FILES, packageChangeOnlyVersion, selectPrProfile, type PrProfileSelection } from './test-pr-profile'; +import { e2eReuseLaneProblem, prepareE2EShardReuse } from './e2e-shard-reuse'; + +type E2EShardReuse = NonNullable>; import { detectBaseBranch, getChangedFiles, @@ -88,7 +91,8 @@ export { PERIODIC_CI_EXCLUDE }; const ROOT = path.resolve(import.meta.dir, '..'); -export type PaidTier = 'gate' | 'periodic'; +export type PaidTier = 'gate' | 'periodic' | 'marathon'; +export const PAID_TIERS: readonly PaidTier[] = ['gate', 'periodic', 'marathon']; export type PaidProfile = 'pr' | 'full'; export interface PaidCaseSelection { @@ -117,6 +121,64 @@ export function isOverlayTestFile(file: string): boolean { return /^skill-e2e-overlay-harness-.+\.test\.ts$/.test(path.basename(normalizeRelativePath(file))); } +/** + * Files whose cases run in separate processes, one shard per registered E2E + * case (`#`): the file's lane wall exceeds one runner's budget + * while every case is short. Separate processes also give each case its own + * SDK semaphore, so shared-libs(-paths) capture waves never queue inside a + * sibling case's wall (the reason paths runs test.serial in one process). + * Every case must be a registered, literal E2E id whose Bun test name is the + * id or its CASE_TEST_NAMES label (test/paid-shards.test.ts scans the sources). + */ +export const CASE_SHARDED_FILES: readonly string[] = [ + 'test/skill-e2e-design.test.ts', + 'test/skill-e2e-plan.test.ts', + 'test/skill-e2e-review-army.test.ts', + 'test/skill-e2e-shared-libs-paths.test.ts', + 'test/skill-e2e-shared-libs.test.ts', +]; + +/** Bun test names that differ from their E2E id. */ +export const CASE_TEST_NAMES: Record = { + 'plan-review-report': '/plan-eng-review writes GSTACK REVIEW REPORT to plan file', + 'auq-format-gate': "/plan-ceo-review's first AskUserQuestion is a compliant decision brief (7/7 + substance)", +}; + +const CASE_KEY_SEPARATOR = '#'; + +/** The test file behind a shard key (`` or `#`). */ +export function shardFile(key: string): string { + return normalizeRelativePath(key).split(CASE_KEY_SEPARATOR)[0]!; +} + +/** The E2E case id of a case shard key, else null. */ +export function shardCaseId(key: string): string | null { + const [, id] = normalizeRelativePath(key).split(CASE_KEY_SEPARATOR); + return id ?? null; +} + +/** Exact Bun name pattern for a set of case ids (labels where the test name differs). */ +export function caseTestNamePattern(ids: string[]): string { + const escaped = ids.map(id => (CASE_TEST_NAMES[id] ?? id).replace(/[.*+?^${}()|[\]\\]/g, '\\$&')); + return `(?:^|\\s)(?:${escaped.join('|')})$`; +} + +/** + * Replace each case-sharded file with one key per registered case of `tier`. + * Throws when such a file's registration is not statically complete: an + * unregistered case would otherwise silently never run. + */ +export function expandCaseShards(files: string[], tier: PaidTier, rootDir = ROOT, + touchfiles: Record = E2E_TOUCHFILES, tiers: Record = E2E_TIERS): string[] { + return files.flatMap(file => { + const rel = normalizeRelativePath(file); + if (!CASE_SHARDED_FILES.includes(rel)) return [file]; + const { registered, known } = fileCaseRegistration(rel, fs.readFileSync(path.join(rootDir, rel), 'utf8'), touchfiles, tiers); + if (!known) throw new Error(`Case-sharded ${rel} needs a complete literal case registration`); + return registered.filter(id => tiers[id] === tier).sort().map(id => `${rel}${CASE_KEY_SEPARATOR}${id}`); + }); +} + /** Compatibility helper for callers that only need the effective wall. */ export function resolvePaidShardTimeoutMs(files: string[], explicitTimeoutMs?: number): number { return resolvePaidShardBudget(files, explicitTimeoutMs).timeoutMs; @@ -157,21 +219,47 @@ export interface TierClassification { * guard runs and self-skips. */ export function classifyPaidTestFile(source: string, tier: PaidTier): TierClassification { - const other: PaidTier = tier === 'gate' ? 'periodic' : 'gate'; const declares = (candidate: PaidTier) => new RegExp(`EVALS_TIER\\s*===\\s*['"\`]${candidate}['"\`]`).test(source) || new RegExp(`\\b(?:describeE2ETier|e2eTierEnabled)\\(\\s*['"\`]${candidate}['"\`]`).test(source); if (declares(tier)) return { included: true, reason: `declares tier '${tier}'` }; - if (declares(other)) return { included: false, reason: `declares tier '${other}' only` }; + const others = PAID_TIERS.filter(candidate => candidate !== tier && declares(candidate)); + if (others.length) return { included: false, reason: `declares tier ${others.map(other => `'${other}'`).join(' and ')} only` }; return { included: true, reason: 'no whole-file tier guard — runtime E2E_TIERS filter decides' }; } +/** + * The E2E ids a paid file registers: the touchfile registrations that list the + * file. `known` is true only when those ids are complete: no computed + * registration (testName, *IfSelected, describeIfSelected with a non-literal + * argument) and every literal registration argument is among them. Quoted + * strings elsewhere (comments, skill paths) never count. + */ +export function fileCaseRegistration( + file: string, source: string, + touchfiles: Record = E2E_TOUCHFILES, + tiers: Record = E2E_TIERS, +): { registered: string[]; known: boolean } { + const rel = normalizeRelativePath(file); + const registered = Object.keys(touchfiles).filter(key => touchfiles[key]!.includes(rel)); + const computed = /testName\s*:\s*(?!string\b)(?:`[^`]*\$\{|[A-Za-z_$])/.test(source) + || /\btest(?:Concurrent)?IfSelected\s*\(\s*(?:`[^`]*\$\{|[A-Za-z_$])/.test(source) + || /\bdescribeIfSelected\s*\([^,]*,(?!\s*\[)/.test(source) + || [...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)].some(m => m[1]!.split(',') + .map(item => item.trim()).some(item => item && !/^(['"`])[^'"`$]*\1$/.test(item))); + const literal = [ + ...[...source.matchAll(/testName\s*:\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!), + ...[...source.matchAll(/\btest(?:Concurrent)?IfSelected\s*\(\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!), + ...[...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)] + .flatMap(m => [...m[1]!.matchAll(/(['"`])([^'"`]+)\1/g)].map(n => n[2]!)), + ].filter(id => id in tiers); + return { registered, known: registered.length > 0 && !computed && literal.every(id => registered.includes(id)) }; +} + /** * A file is skipped for a tier lane only when its registered E2E ids are fully - * known and none of them has that tier. Ids are the touchfile registrations that - * list the file plus literal registration arguments (testName, *IfSelected); - * quoted strings elsewhere (comments, skill paths) never count. Any computed + * known (fileCaseRegistration) and none of them has that tier. Any computed * registration, an id missing from the file's touchfile registration, or no id at * all keeps today's scheduling (the child's runtime filter decides). */ @@ -180,26 +268,27 @@ export function tierSkipReason( touchfiles: Record = E2E_TOUCHFILES, tiers: Record = E2E_TIERS, ): string | null { - const rel = normalizeRelativePath(file); - const registered = Object.keys(touchfiles).filter(key => touchfiles[key]!.includes(rel)); - if (!registered.length) return null; - const computed = /testName\s*:\s*(?:`[^`]*\$\{|[A-Za-z_$])/.test(source) - || /\btest(?:Concurrent)?IfSelected\s*\(\s*(?:`[^`]*\$\{|[A-Za-z_$])/.test(source) - || /\bdescribeIfSelected\s*\([^,]*,(?!\s*\[)/.test(source) - || [...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)].some(m => m[1]!.split(',') - .map(item => item.trim()).some(item => item && !/^(['"`])[^'"`$]*\1$/.test(item))); - if (computed) return null; - const literal = [ - ...[...source.matchAll(/testName\s*:\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!), - ...[...source.matchAll(/\btest(?:Concurrent)?IfSelected\s*\(\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!), - ...[...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)] - .flatMap(m => [...m[1]!.matchAll(/(['"`])([^'"`]+)\1/g)].map(n => n[2]!)), - ].filter(id => id in tiers); - if (literal.some(id => !registered.includes(id))) return null; - if (registered.some(id => tiers[id] === tier)) return null; + const { registered, known } = fileCaseRegistration(file, source, touchfiles, tiers); + if (!known || registered.some(id => tiers[id] === tier)) return null; return `skipped: no E2E_TIERS id has tier ${tier}`; } +/** + * The marathon lane selects positively: a file runs there only when it + * declares the marathon tier or registers a marathon-tier case. Files without + * marathon work never cost a marathon runner, and gate/periodic files never + * gain a third execution. + */ +export function marathonSkipReason( + file: string, source: string, + touchfiles: Record = E2E_TOUCHFILES, + tiers: Record = E2E_TIERS, +): string | null { + if (classifyPaidTestFile(source, 'marathon').reason === "declares tier 'marathon'") return null; + const { registered } = fileCaseRegistration(file, source, touchfiles, tiers); + return registered.some(id => tiers[id] === 'marathon') ? null : 'skipped: declares no marathon tier and registers no marathon case'; +} + export interface TierSelection { selected: string[]; excluded: Array<{ file: string; reason: string }>; @@ -213,11 +302,11 @@ export function selectPaidTestFiles(files: string[], tier: PaidTier, rootDir = R if (carveSkill && files.some(file => carveWrapper(file)) && !files.some(file => carveWrapper(file) === carveSkill)) { throw new Error(`GSTACK_CARVE_SKILL=${carveSkill} has no generic section-loading wrapper`); } - // Periodic-lane exclusions (documented-red / manual-hardware files): a + // Scheduled-lane exclusions (documented-red / manual-hardware files): a // known-red weekly shard is triage waste locally AND in CI, so the list - // applies to every periodic run, with the reason surfaced per file. + // applies to every periodic and marathon run, with the reason surfaced per file. const ciExcluded = (file: string): { reason: string; tracking: string } | undefined => - tier === 'periodic' ? PERIODIC_CI_EXCLUDE[normalizeRelativePath(file)] : undefined; + tier !== 'gate' ? PERIODIC_CI_EXCLUDE[normalizeRelativePath(file)] : undefined; for (const file of files) { // One wrapper per process means a child-side return now creates an empty // shard. Apply the existing explicit cost scope before planning processes. @@ -233,7 +322,8 @@ export function selectPaidTestFiles(files: string[], tier: PaidTier, rootDir = R } const source = fs.readFileSync(path.join(rootDir, file), 'utf8'); const classification = classifyPaidTestFile(source, tier); - const skip = classification.included ? tierSkipReason(file, source, tier) : null; + const skip = !classification.included ? null + : tier === 'marathon' ? marathonSkipReason(file, source) : tierSkipReason(file, source, tier); if (classification.included && !skip) selected.push(file); else excluded.push({ file, reason: skip ?? classification.reason }); } @@ -378,28 +468,29 @@ function packageVersionOnlySinceBase(rootDir: string, baseRef: string): boolean } /** Only audited per-case files, plus the separately selected judge, enter the fast profile. */ +/** The selected PR-profile case ids a shard key owns (a case key owns at most its own case). */ +function prProfileShardIds(key: string, selection: PaidCaseSelection): string[] { + const caseId = shardCaseId(key); + return (PR_PROFILE_FILES[shardFile(key)] ?? []) + .filter(id => (caseId === null || id === caseId) && (selection.e2e === null || selection.e2e.includes(id))); +} + export function prProfileFileSelected(file: string, selection: PaidCaseSelection): boolean { if (file === 'test/skill-llm-eval.test.ts') return selection.judges === null || selection.judges.length > 0; - const ids = PR_PROFILE_FILES[normalizeRelativePath(file)]; - return !!ids && (selection.e2e === null || ids.some(id => selection.e2e!.includes(id))); + return prProfileShardIds(file, selection).length > 0; } export function expectedPrCaseCount(file: string, selection: PaidCaseSelection): number { if (file === 'test/skill-llm-eval.test.ts') return selection.judges?.length ?? Object.keys(LLM_JUDGE_TOUCHFILES).length; - return (PR_PROFILE_FILES[normalizeRelativePath(file)] ?? []).filter(id => selection.e2e === null || selection.e2e.includes(id)).length; + return prProfileShardIds(file, selection).length; } export function prProfileTestNamePattern(file: string, selection: PaidCaseSelection): string { - const labels: Record = { - 'plan-review-report': '/plan-eng-review writes GSTACK REVIEW REPORT to plan file', - 'auq-format-gate': "/plan-ceo-review's first AskUserQuestion is a compliant decision brief (7/7 + substance)", - }; const ids = file === 'test/skill-llm-eval.test.ts' ? selection.judges ?? Object.keys(LLM_JUDGE_TOUCHFILES) - : (PR_PROFILE_FILES[file] ?? []).filter(id => selection.e2e === null || selection.e2e.includes(id)); + : prProfileShardIds(file, selection); if (ids.length === 0) throw new Error(`No selected PR cases for ${file}`); - const escaped = ids.map(id => (labels[id] ?? id).replace(/[.*+?^${}()|[\]\\]/g, '\\$&')); - return `(?:^|\\s)(?:${escaped.join('|')})$`; + return caseTestNamePattern(ids); } export function paidSelectionEnv(profile: PaidProfile, selection: PaidCaseSelection, reason: string): NodeJS.ProcessEnv { @@ -443,6 +534,10 @@ export function diffSkipDecisionForFile( options: DiffSkipOptions = {}, ): ShardSkipDecision { if (selectedNames === null) return { file, kept: true, reason: 'run-all selection' }; + const caseId = shardCaseId(file); + if (caseId !== null) { + return selectedNames.has(caseId) ? { file, kept: true, reason: `selected: ${caseId}` } : { file, kept: false, reason: `case ${caseId} not selected` }; + } const rel = normalizeRelativePath(file); if (!/^test\/skill-e2e-.*\.test\.ts$/.test(rel)) { return { file, kept: true, reason: 'non-skill-e2e paid file — child self-skip authoritative' }; @@ -503,7 +598,7 @@ export function planPaidShards( const shards: string[][] = []; let pending: string[] = []; for (const file of unique) { - if (isOverlayTestFile(file) || FILE_RETRY_BUDGETS.some(budget => budget.file === file)) { + if (isOverlayTestFile(file) || shardCaseId(file) !== null || FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file))) { if (pending.length) shards.push(pending); pending = []; shards.push([file]); @@ -524,7 +619,7 @@ export interface PaidShardBudget { /** Explicit caller limits win; registered supervision preserves existing attempts. */ export function resolvePaidShardBudget(files: string[], overrideMs?: number): PaidShardBudget { - const finding = FILE_RETRY_BUDGETS.find(budget => files.map(normalizeRelativePath).includes(budget.file)); + const finding = FILE_RETRY_BUDGETS.find(budget => files.map(shardFile).includes(budget.file)); if (finding && files.length !== 1) throw new Error('Registered retry budget requires its own shard'); if (overrideMs !== undefined && (!Number.isSafeInteger(overrideMs) || overrideMs <= 0 || overrideMs > 2_147_483_647)) { throw new Error('Shard timeout must be a finite positive timer-safe integer'); @@ -534,8 +629,11 @@ export function resolvePaidShardBudget(files: string[], overrideMs?: number): Pa if (overlay && overrideMs !== undefined && overrideMs < OVERLAY_MIN_FILE_WALL_MS) { throw new Error(`Overlay shard requires at least ${OVERLAY_MIN_FILE_WALL_MS}ms; explicit wall ${overrideMs}ms cannot preserve its work and finalization budget`); } + // A registered file's case shard supervises one case and its allowed attempts. + const registeredMs = finding && shardCaseId(files[0]!) !== null + ? finding.caseMs * (finding.retries + 1) + finding.shardReserveMs : finding?.shardMs; return { - timeoutMs: overrideMs ?? (finding ? finding.shardMs : overlay ? OVERLAY_MIN_FILE_WALL_MS : DEFAULT_SHARD_TIMEOUT_MS), + timeoutMs: overrideMs ?? (registeredMs ?? (overlay ? OVERLAY_MIN_FILE_WALL_MS : DEFAULT_SHARD_TIMEOUT_MS)), source: overrideMs !== undefined ? 'explicit' : finding ? 'registered' : 'default', policyId: finding?.id ?? null, }; @@ -554,8 +652,8 @@ export function buildPaidShardArgs( // Explicit --concurrent/--max-concurrency: the legacy path always set one; // omitting it here made within-shard parallelism differ silently between // the two runners (observed: 1.6x sumdur/wall sharded vs 8x legacy). - // Retries default to 1; RETRY_OVERRIDES membership (old matrix rows' - // earned `retries: 2`) flows through retriesForFiles at the call site. + // Retries come from retriesForFiles (the timeout-is-a-verdict rule) at the + // call site; the fallback of 1 serves only direct callers. return ['test', ...files, '--retry', String(retries ?? 1), '--concurrent', `--max-concurrency=${maxConcurrency}`, `--timeout=${timeoutMs}`]; } @@ -565,7 +663,8 @@ export function buildPaidShardArgs( */ export function shardSlug(files: string[]): string { return files - .map((file) => path.basename(normalizeRelativePath(file)).replace(/\.test\.(?:[cm]?[jt]s|tsx|jsx)$/, '')) + .map((file) => path.basename(shardFile(file)).replace(/\.test\.(?:[cm]?[jt]s|tsx|jsx)$/, '') + + (shardCaseId(file) === null ? '' : `--${shardCaseId(file)}`)) .join('+') .replace(/[^a-zA-Z0-9._+-]/g, '-'); } @@ -599,6 +698,8 @@ export interface ShardOutcome { skippedTests: number | null; /** Effective supervised wall; absent only for unstarted or legacy outcomes. */ budget?: PaidShardBudget; + /** Present when a verified receipt replaced execution (PR lane only). */ + reused?: { inputKey: string; runId: string; revision: string; completedAt: number }; } /** @@ -655,6 +756,10 @@ export interface RunShardsOptions { /** Fast-profile census: selected real cases per file, excluding Bun skips. */ expectedCases?: Record; casePatterns?: Record; + /** The selected case ids per shard key (reported for reused shards). */ + expectedCaseIds?: Record; + /** PR lane only: verified reuse for one shard's exact child environment and wall. */ + reuseFor?: (files: string[], env: NodeJS.ProcessEnv, budget: PaidShardBudget) => E2EShardReuse | null; } let shardLogSequence = 0; @@ -703,17 +808,10 @@ export async function runPaidShard( const log = options.log ?? ((line: string) => console.log(line)); const label = `[test:paid] shard ${shardNumber}/${totalShards}`; - const { command, args } = options.commandFor - ? options.commandFor(files) - : { - command: process.execPath, - args: [...buildPaidShardArgs( - exactTestFileSelectors(files, rootDir), - timeoutMs, - options.withinShardConcurrency ?? DEFAULT_WITHIN_SHARD_CONCURRENCY, - retriesForFiles(files), - ), ...(options.casePatterns ? ['--test-name-pattern', options.casePatterns[files[0]]] : [])], - }; + // A case shard runs exactly its one case; PR patterns narrow further. + const caseId = files.length === 1 ? shardCaseId(files[0]!) : null; + const casePattern = options.casePatterns?.[files[0]!] ?? (caseId !== null ? caseTestNamePattern([caseId]) : undefined); + const expectedCases = options.expectedCases ?? (caseId !== null ? { [files[0]!]: 1 } : undefined); const env = { ...(options.env ?? process.env) }; if (options.evalDirBase) { @@ -725,6 +823,42 @@ export async function runPaidShard( if (!env.GSTACK_CLAUDE_CLI_VERSION) { env.GSTACK_CLAUDE_CLI_VERSION = getClaudeCliVersion(); } + // Verified first-attempt reuse (PR lane only; scripts/e2e-shard-reuse.ts): + // identical consumed inputs to a fresh pass in this PR replace execution + // with an explicitly reported reused result. + // Bootstrap-retention qualification binds per-run state, so that shard stays fresh. + const reuse = files.some(file => normalizeRelativePath(file) === 'test/skill-e2e-qa-workflow.test.ts') + ? null : options.reuseFor?.(files, env, budget) ?? null; + const reused = reuse?.lookup() ?? null; + if (reused) { + const reusedFrom = { input_key: reused.key, run_id: reused.source.runId, revision: reused.source.revision, + completed_at: new Date(reused.source.completedAt).toISOString() }; + const caseIds = options.expectedCaseIds?.[files[0]!] ?? []; + if (env.GSTACK_EVAL_DIR) { + fs.mkdirSync(env.GSTACK_EVAL_DIR, { recursive: true }); + fs.writeFileSync(path.join(env.GSTACK_EVAL_DIR, `e2e-reused-${shardSlug(files)}.json`), `${JSON.stringify({ + schema_version: 1, tier: 'e2e', shard: shardSlug(files), total_tests: caseIds.length, executed_tests: 0, + reused_tests: caseIds.length, passed: caseIds.length, failed: 0, total_cost_usd: 0, total_duration_ms: 0, + tests: caseIds.map(name => ({ name, suite: shardSlug(files), tier: 'e2e', passed: true, duration_ms: 0, cost_usd: 0, + execution: 'reused', reused_from: reusedFrom, attempt: 1 })), + }, null, 2)}\n`); + } + log(`${label} REUSED ${files.join(' ')} — identical inputs passed in run ${reused.source.runId} at ${reusedFrom.completed_at}`); + return { shard: shardNumber, files, status: 'passed', exitCode: 0, elapsedMs: 0, groupPid: null, + executedTests: caseIds.length, skippedTests: 0, budget, + reused: { inputKey: reused.key, runId: reused.source.runId, revision: reused.source.revision, completedAt: reused.source.completedAt } }; + } + const { command, args } = options.commandFor + ? options.commandFor(files) + : { + command: process.execPath, + args: [...buildPaidShardArgs( + exactTestFileSelectors(files.map(shardFile), rootDir), + timeoutMs, + options.withinShardConcurrency ?? DEFAULT_WITHIN_SHARD_CONCURRENCY, + retriesForFiles(files), + ), ...(casePattern !== undefined ? ['--test-name-pattern', casePattern] : [])], + }; // Per-shard temp + Chromium-profile isolation — the free runner treats // this as mandatory (test-free-shards.ts: two concurrent shards on one // profile dir kill each other's browser; shared tmp cross-contaminates), @@ -881,15 +1015,16 @@ export async function runPaidShard( let status: ShardStatus = timedOut ? 'timed-out' : !retentionFailed && !logWriteFailed && !incompleteCapture && strictTestExitCode(exitCode ?? 1, summary, expectedFiles) === 0 ? 'passed' : 'failed'; - if (status === 'passed' && options.expectedCases) { - const expected = files.reduce((count, file) => count + (options.expectedCases![file] ?? 0), 0); + if (status === 'passed' && expectedCases) { + const expected = files.reduce((count, file) => count + (expectedCases[file] ?? 0), 0); const actual = summary.terminalTestCounts.reduce((count, value) => count + value, 0) - summary.skippedTests; if (expected < 1 || actual !== expected) { status = 'failed'; - log(`${label} expected ${expected} selected cases, executed ${actual}; refusing incomplete PR coverage`); + log(`${label} expected ${expected} selected cases, executed ${actual}; refusing incomplete case coverage`); } } const elapsedMs = Date.now() - startedAt; + if (status === 'passed' && reuse) reuse.publish(); // Failure debuggability without the RAM cost: read back only the log's // tail. Live mode already streamed everything, so no re-print there. @@ -1077,6 +1212,16 @@ export interface ManifestEntry { reason?: string; /** Required when a registered retry-budget file is planned. */ budget?: PaidShardBudget; + /** Budget-mode packing weight (recorded wall, or the whole budget when unknown). */ + estimatedMs?: number; +} + +/** Budget-mode plan: per-executor estimate and the CI job timeout it needs. */ +export interface PaidSlicePlan { + sliceBudgetMs: number; + jobs: number; + estimatedSliceMs: number[]; + ciTimeoutMinutes: number; } export interface PaidRunManifest { @@ -1089,43 +1234,57 @@ export interface PaidRunManifest { profile?: PaidProfile; selection?: PaidCaseSelection; prCoverage?: PrProfileSelection; + plan?: PaidSlicePlan; entries: ManifestEntry[]; } /** - * Files whose old evals.yml matrix rows carried `retries: 2`, with the - * receipts that earned them (see the deleted rows' comments). The runner - * default stays --retry 1; membership here is a literals map so retry - * parity with the matrix is explicit, not folklore. + * Automatic retries follow the approved rule in test/helpers/eval-budgets.ts: + * a timed-out attempt is a verdict, so only files whose every case budget is + * at most RETRY_MAX_CASE_MS keep a retry (registered rows derive it from their + * caseMs; SHORT_CASE_RETRY_FILES lists the rest). Overlays and every other + * paid file run once. A multi-file shard takes the smallest allowance. */ -export const RETRY_OVERRIDES: Record = { - 'test/skill-e2e-workflow.test.ts': 2, - 'test/skill-e2e-office-hours-auto-mode.test.ts': 2, - 'test/skill-e2e-plan-mode-no-op.test.ts': 2, -}; - export function retriesForFiles(files: string[]): number { if (files.some(isOverlayTestFile)) return 0; - return Math.max(1, ...files.map((f) => RETRY_OVERRIDES[normalizeRelativePath(f)] ?? 1)); + return Math.min(...files.map((file) => { + const rel = shardFile(file); + const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === rel); + if (registered) return registered.retries; + return SHORT_CASE_RETRY_FILES.includes(rel) ? 1 : 0; + })); } export const PAID_TEST_DURATIONS_FILE = 'scripts/paid-test-durations.json'; /** - * Recorded per-file paid-shard wall times (ms) from real CI slice reports, - * refreshed with `--report --write-durations`. A packing hint only: a - * missing or corrupt seed keeps the supervision-budget allocation. + * Recorded per-file paid-shard wall times (ms) per tier from real CI slice + * reports (a file's gate and periodic cases differ), refreshed with + * `--report --write-durations`. A packing hint only: a missing or + * corrupt seed keeps the supervision-budget allocation. */ -export function loadPaidTestDurations(rootDir = ROOT): Record { +export function loadPaidTestDurations(rootDir = ROOT, tier: PaidTier = 'gate'): Record { try { - const parsed = JSON.parse(fs.readFileSync(path.join(rootDir, PAID_TEST_DURATIONS_FILE), 'utf8')) as { durations?: Record }; - return Object.fromEntries(Object.entries(parsed.durations ?? {}) + const parsed = JSON.parse(fs.readFileSync(path.join(rootDir, PAID_TEST_DURATIONS_FILE), 'utf8')) as { version?: unknown; tiers?: Record> }; + if (parsed.version !== 2) return {}; + return Object.fromEntries(Object.entries(parsed.tiers?.[tier] ?? {}) .filter((entry): entry is [string, number] => typeof entry[1] === 'number' && Number.isFinite(entry[1]) && entry[1] > 0)); } catch { return {}; } } +/** Rewrite one tier of the committed seed, keeping the other tiers. */ +export function writePaidTestDurations(tier: PaidTier, durations: Record, rootDir = ROOT): void { + const target = path.join(rootDir, PAID_TEST_DURATIONS_FILE); + const tiers = Object.fromEntries(PAID_TIERS.map(name => [name, loadPaidTestDurations(rootDir, name)]) + .filter(([name, recorded]) => name === tier || Object.keys(recorded as object).length > 0)); + tiers[tier] = durations; + const temporary = `${target}.tmp-${process.pid}`; + fs.writeFileSync(temporary, `${JSON.stringify({ version: 2, recordedAt: new Date().toISOString(), tiers }, null, 2)}\n`); + fs.renameSync(temporary, target); +} + /** Merge a report's executed single-file outcomes into the seed; all-skipped shards carry no cost signal. */ export function mergePaidTestDurations(seed: Record, results: SliceResult[]): Record { const merged = { ...seed }; @@ -1138,6 +1297,65 @@ export function mergePaidTestDurations(seed: Record, results: Sl return Object.fromEntries(Object.entries(merged).sort(([a], [b]) => (a < b ? -1 : 1))); } +/** Setup, image pull and artifact upload allowance on top of a slice's supervised wall. */ +export const CI_SETUP_ALLOWANCE_MINUTES = 20; + +/** Estimated wall of one executor running `files` in order on `jobs` FIFO workers. */ +export function estimatedSliceMs(files: string[], weight: (file: string) => number, jobs: number): number { + const workers = Array(Math.max(1, jobs)).fill(0); + for (const file of files) { + const next = workers.indexOf(Math.min(...workers)); + workers[next] += weight(file); + } + return Math.max(...workers); +} + +/** Executor order within a slice: longest recorded work first, then by path. */ +export function sliceExecutionOrder(entries: T[]): T[] { + return [...entries].sort((a, b) => (b.estimatedMs ?? 0) - (a.estimatedMs ?? 0) || (a.file < b.file ? -1 : a.file > b.file ? 1 : 0)); +} + +/** Supervised worst case of one slice in execution order, overlays at their own admission limit. */ +export function sliceSupervisedWallMs(files: string[], jobs: number, overrideMs?: number): number { + return paidShardWallUpperBoundMs(files.filter(file => !isOverlayTestFile(file)), jobs, overrideMs) + + paidShardWallUpperBoundMs(files.filter(isOverlayTestFile), Math.min(jobs, OVERLAY_MAX_ACTIVE_SHARDS), overrideMs); +} + +/** + * Budget packing: one runner per file or per tightly packed group. Files go + * longest-recorded-first into the fullest slice whose estimated wall stays + * within the budget (best fit), else into a new slice. A file with no recorded + * wall weighs the whole budget, so unknown cost gets a runner of its own. A + * file longer than the budget runs alone. Overlay wrappers keep one shared + * final slice (one wrapper at a time). The CI timeout covers every slice's + * supervised worst case plus the setup allowance. + */ +export function packBySliceBudget(files: string[], budgetMs: number, jobs: number, + recorded: Record, timeoutMs?: number): { + slices: string[][]; estimates: Record; estimatedSliceMs: number[]; ciTimeoutMinutes: number; +} { + const estimates = Object.fromEntries(files.map(file => [file, recorded[normalizeRelativePath(file)] ?? budgetMs])); + const weight = (file: string) => estimates[file]!; + const slices: string[][] = []; + for (const file of sliceExecutionOrder(files.filter(file => !isOverlayTestFile(file)).map(file => ({ file, estimatedMs: weight(file) }))).map(entry => entry.file)) { + let best = -1, bestMs = -1; + slices.forEach((planned, index) => { + const ms = estimatedSliceMs([...planned, file], weight, jobs); + if (ms <= budgetMs && ms > bestMs) { best = index; bestMs = ms; } + }); + if (best < 0) slices.push([file]); + else slices[best]!.push(file); + } + const overlays = files.filter(isOverlayTestFile).sort(); + if (overlays.length) slices.push(sliceExecutionOrder(overlays.map(file => ({ file, estimatedMs: weight(file) }))).map(entry => entry.file)); + if (!slices.length) slices.push([]); + const estimatedSliceMsList = slices.map(planned => planned.some(isOverlayTestFile) + ? estimatedSliceMs(planned, weight, Math.min(jobs, OVERLAY_MAX_ACTIVE_SHARDS)) : estimatedSliceMs(planned, weight, jobs)); + const worst = Math.max(0, ...slices.map(planned => sliceSupervisedWallMs(planned, jobs, timeoutMs))); + return { slices, estimates, estimatedSliceMs: estimatedSliceMsList, + ciTimeoutMinutes: Math.ceil(worst / 60_000) + CI_SETUP_ALLOWANCE_MINUTES }; +} + /** Worker counts whose worst-case slice wall duration packing may never worsen. */ export const SUPERVISED_WORKER_COUNTS = [1, 2, 3, 4] as const; @@ -1153,7 +1371,13 @@ export const SUPERVISED_WORKER_COUNTS = [1, 2, 3, 4] as const; export function buildRunManifest(opts: { tier: PaidTier; profile?: PaidProfile; - sliceCount: number; + /** Fixed slice count; exclusive with sliceBudgetMs. */ + sliceCount?: number; + /** Budget mode: pack recorded work so each executor's estimated wall stays + * within this budget; the slice count follows from the plan. */ + sliceBudgetMs?: number; + /** Shard workers per executor (EVALS_JOBS) that budget mode plans for. */ + jobs?: number; evalsAll: boolean; timeoutMs?: number; discovered?: string[]; @@ -1165,7 +1389,12 @@ export function buildRunManifest(opts: { /** Weekly gate census only: LLM judges already run in the periodic census and PR gate lanes. */ skipJudges?: boolean; }): PaidRunManifest { - if (!Number.isInteger(opts.sliceCount) || opts.sliceCount <= 0) { + const budgetMode = opts.sliceBudgetMs !== undefined; + if (budgetMode === (opts.sliceCount !== undefined)) throw new Error('Plan with exactly one of --slices or --slice-budget'); + if (budgetMode && (!Number.isSafeInteger(opts.sliceBudgetMs) || opts.sliceBudgetMs! <= 0 || !Number.isSafeInteger(opts.jobs) || opts.jobs! <= 0)) { + throw new Error('--slice-budget needs a positive budget and an explicit positive --jobs'); + } + if (!budgetMode && (!Number.isInteger(opts.sliceCount) || opts.sliceCount! <= 0)) { throw new Error(`--slices needs a positive integer. Received: ${opts.sliceCount}`); } const rootDir = opts.rootDir ?? ROOT; @@ -1178,7 +1407,7 @@ export function buildRunManifest(opts: { const selected = opts.skipJudges ? tierSelection.selected.filter(file => !judge(file)) : tierSelection.selected; const excluded = [...tierSelection.excluded, ...(opts.skipJudges ? tierSelection.selected.filter(judge) .map(file => ({ file, reason: 'skipped: LLM judges run in the periodic census and PR gate lanes' })) : [])]; - const shards = planPaidShards(selected, { maxFilesPerShard: 1 }); + const shards = planPaidShards(expandCaseShards(selected, opts.tier, rootDir), { maxFilesPerShard: 1 }); const cases = computePaidCaseSelection({ profile, env, rootDir, changedFiles: opts.changedFiles }); const fast = cases.coverage?.mode === 'pr'; const profileShards = fast ? shards.filter(files => prProfileFileSelected(files[0], cases.selection)) : shards; @@ -1189,14 +1418,32 @@ export function buildRunManifest(opts: { } const entries: ManifestEntry[] = []; - const overlaySlice = opts.sliceCount; + if (budgetMode) { + const plan = packBySliceBudget(runnable.map(files => files[0]!), opts.sliceBudgetMs!, opts.jobs!, + opts.durations ?? loadPaidTestDurations(rootDir, opts.tier), opts.timeoutMs); + plan.slices.forEach((files, index) => files.forEach(file => entries.push({ file, slice: index + 1, status: 'planned', + estimatedMs: plan.estimates[file]!, + ...(FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file)) ? { budget: resolvePaidShardBudget([file], opts.timeoutMs) } : {}) }))); + for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason }); + for (const e of excluded) entries.push({ file: e.file, slice: 0, status: 'excluded', reason: e.reason }); + entries.sort((a, b) => (a.file < b.file ? -1 : 1)); + return parseRunManifest(JSON.stringify({ + version: 1, tier: opts.tier, evalsAll: opts.evalsAll, sliceCount: plan.slices.length, + selectionReason: cases.reason, profile, selection: cases.selection, + ...(cases.coverage ? { prCoverage: cases.coverage } : {}), + plan: { sliceBudgetMs: opts.sliceBudgetMs!, jobs: opts.jobs!, estimatedSliceMs: plan.estimatedSliceMs, ciTimeoutMinutes: plan.ciTimeoutMinutes }, + entries, + } satisfies PaidRunManifest)); + } + const sliceCount = opts.sliceCount!; + const overlaySlice = sliceCount; const reserveOverlaySlice = overlaySlice > 1 && runnable.some(files => files.some(isOverlayTestFile)); const ordinarySlices = overlaySlice - Number(reserveOverlaySlice); // Spread registered long files by supervised load. Keep one ordinary-only // lane when possible, so every lane does not inherit a long-workflow tail. // The reserved overlay slice retains its ownership. const ordinary = runnable.filter(files => !files.some(isOverlayTestFile)); - const registered = ordinary.filter(files => FILE_RETRY_BUDGETS.some(budget => budget.file === files[0])); + const registered = ordinary.filter(files => FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(files[0]!))); const allocations = new Map(); if (registered.length && ordinarySlices > 1) { const loads = Array(ordinarySlices).fill(0); @@ -1220,7 +1467,7 @@ export function buildRunManifest(opts: { } const packed = packByRecordedDuration(); function packByRecordedDuration(): Map | null { - const recorded = opts.durations ?? loadPaidTestDurations(rootDir); + const recorded = opts.durations ?? loadPaidTestDurations(rootDir, opts.tier); if (ordinarySlices < 2 || ordinary.length === 0 || Object.keys(recorded).length === 0) return null; const bound = (files: string[], jobs: number) => paidShardWallUpperBoundMs([...files].sort(), jobs, opts.timeoutMs); const lanes = Array.from({ length: ordinarySlices }, (_, lane) => @@ -1267,7 +1514,7 @@ export function buildRunManifest(opts: { runnable.forEach((files) => { const slice = files.some(isOverlayTestFile) ? overlaySlice : (packed ?? allocations).get(files[0])!; entries.push({ file: files[0], slice, status: 'planned', - ...(FILE_RETRY_BUDGETS.some(budget => budget.file === files[0]) + ...(FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(files[0]!)) ? { budget: resolvePaidShardBudget(files, opts.timeoutMs) } : {}) }); }); for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason }); @@ -1278,7 +1525,7 @@ export function buildRunManifest(opts: { version: 1, tier: opts.tier, evalsAll: opts.evalsAll, - sliceCount: opts.sliceCount, + sliceCount, selectionReason: cases.reason, profile, selection: cases.selection, @@ -1288,10 +1535,25 @@ export function buildRunManifest(opts: { return parseRunManifest(JSON.stringify(manifest)); } +/** + * Narrow a built manifest to a curated case subset (validation phases): the + * selection binds every child, and a case shard outside it can execute + * nothing, so it becomes skipped instead of an empty planned shard. + */ +export function restrictManifestSelection(manifest: PaidRunManifest, selection: PaidCaseSelection, reason: string): PaidRunManifest { + const entries = manifest.entries.map(entry => { + const caseId = shardCaseId(entry.file); + if (entry.status !== 'planned' || caseId === null || selection.e2e === null || selection.e2e.includes(caseId)) return entry; + const { estimatedMs: _estimate, budget: _budget, ...rest } = entry; + return { ...rest, slice: 0, status: 'skipped-by-diff' as const, reason }; + }); + return parseRunManifest(JSON.stringify({ ...manifest, selection, entries })); +} + export function parseRunManifest(raw: string): PaidRunManifest { const parsed = JSON.parse(raw) as PaidRunManifest; if (parsed.version !== 1) throw new Error(`unsupported manifest version: ${(parsed as { version?: unknown }).version}`); - if (parsed.tier !== 'gate' && parsed.tier !== 'periodic') throw new Error(`manifest tier invalid: ${parsed.tier}`); + if (!PAID_TIERS.includes(parsed.tier)) throw new Error(`manifest tier invalid: ${parsed.tier}`); if (parsed.profile !== undefined && parsed.profile !== 'pr' && parsed.profile !== 'full') throw new Error('manifest profile invalid'); if (parsed.selection !== undefined) { for (const [key, inventory] of [['e2e', E2E_TOUCHFILES], ['judges', LLM_JUDGE_TOUCHFILES]] as const) { @@ -1329,19 +1591,50 @@ export function parseRunManifest(raw: string): PaidRunManifest { if (entry.status === 'planned' && (entry.slice < 1 || entry.slice > parsed.sliceCount)) { throw new Error(`planned entry ${entry.file} has out-of-range slice ${entry.slice}`); } + if (entry.estimatedMs !== undefined && (entry.status !== 'planned' || !Number.isSafeInteger(entry.estimatedMs) || entry.estimatedMs < 0)) { + throw new Error(`manifest entry ${entry.file} has an invalid estimate`); + } if (entry.status === 'planned' && parsed.prCoverage?.mode === 'pr' && !prProfileFileSelected(entry.file, parsed.selection!)) { throw new Error(`manifest file is outside its PR case selection: ${entry.file}`); } } + for (const entry of parsed.entries) { + const caseId = shardCaseId(entry.file); + if (caseId === null ? entry.status === 'planned' && CASE_SHARDED_FILES.includes(shardFile(entry.file)) + : !CASE_SHARDED_FILES.includes(shardFile(entry.file)) || !(caseId in E2E_TOUCHFILES)) { + throw new Error(`Case-sharded files plan one registered case per shard: ${entry.file}`); + } + if (caseId !== null && entry.status === 'planned' && parsed.selection?.e2e && !parsed.selection.e2e.includes(caseId)) { + throw new Error(`Planned case shard is outside the manifest selection: ${entry.file}`); + } + } if (parsed.prCoverage?.mode === 'pr') { - const required = Object.entries(PR_PROFILE_FILES).filter(([, ids]) => ids.some(id => parsed.selection!.e2e!.includes(id))).map(([file]) => file); - if (parsed.selection!.judges!.length) required.push('test/skill-llm-eval.test.ts'); - for (const file of required) { - if (parsed.entries.filter(entry => entry.file === file && entry.status === 'planned').length !== 1) { - throw new Error(`PR selected cases require exactly one planned owning file: ${file}`); + const planned = parsed.entries.filter(entry => entry.status === 'planned').map(entry => normalizeRelativePath(entry.file)); + const required: string[][] = Object.entries(PR_PROFILE_FILES).flatMap(([file, ids]) => { + const selected = ids.filter(id => parsed.selection!.e2e!.includes(id)); + if (!selected.length) return []; + return [CASE_SHARDED_FILES.includes(file) ? selected.map(id => `${file}#${id}`) : [file]]; + }); + if (parsed.selection!.judges!.length) required.push(['test/skill-llm-eval.test.ts']); + for (const owners of required) { + if (owners.some(owner => planned.filter(key => key === owner).length !== 1) + || planned.filter(key => shardFile(key) === shardFile(owners[0]!)).length !== owners.length) { + throw new Error(`PR selected cases require exactly one planned owning file: ${owners.join(', ')}`); } } } + if (parsed.plan !== undefined) { + const plan = parsed.plan; + const count = (value: unknown) => Number.isSafeInteger(value) && Number(value) > 0; + if (!plan || typeof plan !== 'object' || !count(plan.sliceBudgetMs) || !count(plan.jobs) || !count(plan.ciTimeoutMinutes) + || !Array.isArray(plan.estimatedSliceMs) || plan.estimatedSliceMs.length !== parsed.sliceCount + || !plan.estimatedSliceMs.every(ms => Number.isSafeInteger(ms) && ms >= 0) + || parsed.entries.some(entry => entry.status === 'planned' && entry.estimatedMs === undefined)) { + throw new Error('manifest slice plan malformed'); + } + } + const keys = parsed.entries.map(entry => normalizeRelativePath(entry.file)); + if (new Set(keys).size !== keys.length) throw new Error('Duplicate manifest entry'); const overlaySlice = parsed.sliceCount; const plannedOverlays = parsed.entries.filter(entry => entry.status === 'planned' && isOverlayTestFile(entry.file)); if (plannedOverlays.some(entry => entry.slice !== overlaySlice)) { @@ -1352,8 +1645,8 @@ export function parseRunManifest(raw: string): PaidRunManifest { throw new Error('The final ordinary manifest slice is reserved for overlay files'); } for (const budget of FILE_RETRY_BUDGETS) { - const entries = parsed.entries.filter(entry => normalizeRelativePath(entry.file) === budget.file); - if (entries.length > 1) throw new Error(`Duplicate registered manifest entry: ${budget.file}`); + const entries = parsed.entries.filter(entry => shardFile(entry.file) === budget.file); + if (entries.length > 1 && entries.some(entry => shardCaseId(entry.file) === null)) throw new Error(`Duplicate registered manifest entry: ${budget.file}`); for (const entry of entries.filter(entry => entry.status === 'planned')) { if (!entry.budget) throw new Error(`Registered manifest needs an explicit budget record: ${budget.file}`); const expected = resolvePaidShardBudget([entry.file], entry.budget.source === 'explicit' ? entry.budget.timeoutMs : undefined); @@ -1371,7 +1664,7 @@ export interface SliceResult { sliceIndex: number; sliceCount: number; timeoutOverrideMs?: number; - outcomes: Array>; + outcomes: Array>; } /** @@ -1405,7 +1698,7 @@ export function verifySliceResults( const reported = new Map(); for (const result of results) { for (const outcome of result.outcomes) { - if (outcome.files.some(file => FILE_RETRY_BUDGETS.some(budget => budget.file === normalizeRelativePath(file))) && outcome.files.length !== 1) { + if (outcome.files.some(file => FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file)) || shardCaseId(file) !== null) && outcome.files.length !== 1) { problems.push('Registered result must report its own shard'); } const file = normalizeRelativePath(outcome.files[0] ?? ''); @@ -1417,9 +1710,21 @@ export function verifySliceResults( problems.push(`PR profile expected ${expected} executed cases in ${file}, received ${executed}`); } } + if (shardCaseId(file) !== null && outcome.status === 'passed' + && (outcome.executedTests === null || outcome.skippedTests === null || outcome.executedTests - outcome.skippedTests !== 1)) { + problems.push(`Case shard must execute exactly its one case: ${file}`); + } + if (outcome.reused !== undefined) { + const r = outcome.reused; + if (manifest.prCoverage?.mode !== 'pr') problems.push(`${file}: only the fast PR profile may reuse results; this lane executes fresh`); + if (outcome.status !== 'passed' || outcome.exitCode !== 0 || !/^[a-f0-9]{64}$/.test(r?.inputKey ?? '') + || !/^[\w./-]{1,160}$/.test(r?.runId ?? '') || !/^[a-f0-9]{40}$/.test(r?.revision ?? '') || !Number.isSafeInteger(r?.completedAt) || r.completedAt <= 0) { + problems.push(`${file}: malformed reused result`); + } + } if (reported.has(file)) problems.push(`${file} reported by two slices`); reported.set(file, { slice: result.sliceIndex, status: outcome.status }); - const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === file); + const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === shardFile(file)); const finding = STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === file); if (finding) { // Full-census runs must account for every registered case. A manifest @@ -1453,6 +1758,22 @@ export function verifySliceResults( return { ok: problems.length === 0, problems }; } +/** Budget-mode plan lines: every slice with its estimate, files and retries. */ +export function formatSlicePlan(manifest: PaidRunManifest): string[] { + const plan = manifest.plan; + if (!plan) return []; + const minutes = (ms: number) => (ms / 60_000).toFixed(1); + const lines = [`[test:paid] slice plan: ${manifest.sliceCount} slice(s) x ${plan.jobs} worker(s), budget ${minutes(plan.sliceBudgetMs)}m per slice, ` + + `longest estimate ${minutes(Math.max(0, ...plan.estimatedSliceMs))}m, CI job timeout ${plan.ciTimeoutMinutes}m`]; + for (let slice = 1; slice <= manifest.sliceCount; slice++) { + const mine = sliceExecutionOrder(manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice)); + const over = plan.estimatedSliceMs[slice - 1]! > plan.sliceBudgetMs ? ' [over budget: longer than one runner allows]' : ''; + lines.push(` slice ${slice}: ~${minutes(plan.estimatedSliceMs[slice - 1]!)}m${over}`); + for (const entry of mine) lines.push(` ${entry.file} ~${minutes(entry.estimatedMs ?? 0)}m retries=${retriesForFiles([entry.file])}`); + } + return lines; +} + export function formatProfileCoverage(manifest: PaidRunManifest): string[] { const coverage = manifest.prCoverage; return [ @@ -1497,6 +1818,9 @@ type CliOptions = { emitPlanPath: string | null; /** Slice count for --emit-plan. */ slices: number; + /** Budget mode for --emit-plan / --list: per-executor estimated wall. */ + sliceBudgetMs: number | null; + jobsExplicit: boolean; /** Executor mode: consume this manifest... */ planPath: string | null; /** ...running only this 1-based slice. */ @@ -1519,10 +1843,10 @@ function validatedTier(value: string | undefined, source: string): PaidTier { // otherwise cast through unchecked, match nothing in the runtime E2E_TIERS // filter, self-skip every test, and exit 0 with all shards 'passed' — the // exact 0%-execution-looks-like-a-pass class this runner exists to kill. - if (value !== 'gate' && value !== 'periodic') { - throw new Error(`${source} must be gate or periodic. Received: ${value}`); + if (!PAID_TIERS.includes(value as PaidTier)) { + throw new Error(`${source} must be gate, periodic or marathon. Received: ${value}`); } - return value; + return value as PaidTier; } function validatedProfile(value: string | undefined, source: string): PaidProfile { @@ -1553,6 +1877,8 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process maxFilesPerShard: DEFAULT_MAX_FILES_PER_SHARD, emitPlanPath: null, slices: 1, + sliceBudgetMs: null, + jobsExplicit: !!env.EVALS_JOBS, planPath: null, sliceIndex: null, reportDir: null, @@ -1563,9 +1889,7 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process const arg = argv[index]; if (arg === '--list') { options.listOnly = true; continue; } if (arg === '--tier') { - const value = argv[index += 1]; - if (value !== 'gate' && value !== 'periodic') throw new Error(`--tier must be gate or periodic. Received: ${value}`); - options.tier = value; + options.tier = validatedTier(argv[index += 1] ?? '-', '--tier'); continue; } if (arg === '--profile') { @@ -1574,7 +1898,8 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process options.profile = validatedProfile(value, '--profile'); options.profileExplicit = true; continue; } if (arg === '--timeout') { options.timeoutMs = parsePositiveInt(argv[index += 1], '--timeout') * 1000; options.timeoutExplicit = true; continue; } - if (arg === '--jobs') { options.jobs = parsePositiveInt(argv[index += 1], '--jobs'); continue; } + if (arg === '--jobs') { options.jobs = parsePositiveInt(argv[index += 1], '--jobs'); options.jobsExplicit = true; continue; } + if (arg === '--slice-budget') { options.sliceBudgetMs = parsePositiveInt(argv[index += 1], '--slice-budget') * 1000; continue; } if (arg === '--files-per-shard') { options.maxFilesPerShard = parsePositiveInt(argv[index += 1], '--files-per-shard'); continue; } if (arg === '--emit-plan') { const value = argv[index += 1]; @@ -1598,6 +1923,8 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process throw new Error(`Unknown argument: ${arg}`); } if (options.writeDurations && !options.reportDir) throw new Error('--write-durations requires --report'); + if (options.sliceBudgetMs !== null && argv.includes('--slices')) throw new Error('Plan with exactly one of --slices or --slice-budget'); + if (options.sliceBudgetMs !== null && !options.jobsExplicit) throw new Error('--slice-budget needs explicit --jobs (or EVALS_JOBS): the plan packs and supervises for that worker count'); if (options.skipJudges && (!options.emitPlanPath || options.tier !== 'gate')) throw new Error('--skip-judges applies only to an emitted gate census plan'); if (options.profile === 'pr' && options.tier !== 'gate') throw new Error('PR profile requires gate tier'); if (options.profile === 'pr' && options.maxFilesPerShard !== 1) throw new Error('PR profile requires one file per shard to preserve case accounting'); @@ -1613,7 +1940,7 @@ async function main(): Promise { const manifest = buildRunManifest({ tier: options.tier, profile: options.profile, - sliceCount: options.slices, + ...(options.sliceBudgetMs !== null ? { sliceBudgetMs: options.sliceBudgetMs, jobs: options.jobs } : { sliceCount: options.slices }), timeoutMs: options.timeoutExplicit ? options.timeoutMs : undefined, evalsAll: process.env.EVALS_ALL === '1', skipJudges: options.skipJudges, @@ -1628,6 +1955,7 @@ async function main(): Promise { + `${planned} planned across ${manifest.sliceCount} slice(s), ${skipped} skipped by diff, ` + `${excludedCount} excluded (${manifest.selectionReason})`, ); + for (const line of formatSlicePlan(manifest)) console.log(line); return 0; } @@ -1647,16 +1975,14 @@ async function main(): Promise { for (const line of formatProfileCoverage(manifest)) console.log(line); for (const result of results.sort((a, b) => a.sliceIndex - b.sliceIndex)) { for (const outcome of result.outcomes) { - console.log(` slice ${result.sliceIndex} ${outcome.status.padEnd(15)} ${String(Math.round(outcome.elapsedMs / 1000)).padStart(5)}s ${outcome.files.join(' ')}`); + const shown = outcome.reused ? `reused (run ${outcome.reused.runId})` : outcome.status; + console.log(` slice ${result.sliceIndex} ${shown.padEnd(15)} ${String(Math.round(outcome.elapsedMs / 1000)).padStart(5)}s ${outcome.files.join(' ')}`); } } if (options.writeDurations) { - const durations = mergePaidTestDurations(loadPaidTestDurations(), results); - const target = path.join(ROOT, PAID_TEST_DURATIONS_FILE); - const temporary = `${target}.tmp-${process.pid}`; - fs.writeFileSync(temporary, `${JSON.stringify({ version: 1, recordedAt: new Date().toISOString(), durations }, null, 2)}\n`); - fs.renameSync(temporary, target); - console.log(`[test:paid] wrote ${Object.keys(durations).length} durations to ${PAID_TEST_DURATIONS_FILE}`); + const durations = mergePaidTestDurations(loadPaidTestDurations(ROOT, manifest.tier), results); + writePaidTestDurations(manifest.tier, durations); + console.log(`[test:paid] wrote ${Object.keys(durations).length} ${manifest.tier} durations to ${PAID_TEST_DURATIONS_FILE}`); } // Historical flaky_retries includes every case with multiple attempts, // whether its final result passed or failed. Report attempts separately @@ -1767,7 +2093,10 @@ async function main(): Promise { if (options.sliceIndex > manifest.sliceCount) { throw new Error(`--slice ${options.sliceIndex} exceeds manifest sliceCount ${manifest.sliceCount}`); } - const mine = manifest.entries.filter((e) => e.status === 'planned' && e.slice === options.sliceIndex); + if (manifest.plan && options.jobs !== manifest.plan.jobs) { + throw new Error(`manifest was packed for ${manifest.plan.jobs} worker(s) per slice; EVALS_JOBS=${options.jobs} would break its supervision bound`); + } + const mine = sliceExecutionOrder(manifest.entries.filter((e) => e.status === 'planned' && e.slice === options.sliceIndex)); const shards = mine.map((e) => [e.file]); for (const files of shards) resolvePaidShardTimeoutMs(files, timeoutOverride); console.log(`[test:paid] slice ${options.sliceIndex}/${manifest.sliceCount}: ${shards.length} shard(s), tier=${manifest.tier}, evalsAll=${manifest.evalsAll}`); @@ -1775,7 +2104,7 @@ async function main(): Promise { if (options.listOnly) { for (const [index, files] of shards.entries()) { const budget = resolvePaidShardBudget(files, timeoutOverride); - console.log(` shard ${index + 1}/${shards.length}: ${files.join(' ')} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'}`); + console.log(` shard ${index + 1}/${shards.length}: ${files.join(' ')} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'} retries=${retriesForFiles(files)}`); } return 0; } @@ -1794,6 +2123,18 @@ async function main(): Promise { ...(manifest.prCoverage?.mode === 'pr' ? { expectedCases: Object.fromEntries(mine.map(entry => [entry.file, expectedPrCaseCount(entry.file, manifest.selection!)])), casePatterns: Object.fromEntries(mine.map(entry => [entry.file, prProfileTestNamePattern(entry.file, manifest.selection!)])), + expectedCaseIds: Object.fromEntries(mine.map(entry => [entry.file, prProfileShardIds(entry.file, manifest.selection!)])), + reuseFor: e2eReuseLaneProblem(process.env, manifest.prCoverage.mode) !== null ? undefined : (files, env, budget) => { + const key = files[0]!; + const file = shardFile(key); + if (files.length !== 1 || !/^test\/skill-e2e-/.test(file)) return null; + const { registered, known } = fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')); + return prepareE2EShardReuse({ root: ROOT, key, file, caseIds: prProfileShardIds(key, manifest.selection!), + registeredIds: registered, registrationKnown: known, + casePattern: prProfileTestNamePattern(key, manifest.selection!), expectedCases: expectedPrCaseCount(key, manifest.selection!), + retries: retriesForFiles(files), timeoutMs: budget.timeoutMs, withinShardConcurrency: options.withinShardConcurrency, + tier: manifest.tier, profile, env }); + }, } : {}), env: { ...process.env, @@ -1821,8 +2162,8 @@ async function main(): Promise { sliceIndex: options.sliceIndex, sliceCount: manifest.sliceCount, ...(options.timeoutExplicit ? { timeoutOverrideMs: options.timeoutMs } : {}), - outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget }) => - ({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}) })), + outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused }) => + ({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}), ...(reused ? { reused } : {}) })), }; fs.mkdirSync(evalDirBase, { recursive: true }); const sliceResultPath = path.join(evalDirBase, `slice-${options.sliceIndex}.json`); @@ -1832,8 +2173,16 @@ async function main(): Promise { return summaryExitCode(summary); } + if (options.listOnly && options.sliceBudgetMs !== null) { + const manifest = buildRunManifest({ tier: options.tier, profile: options.profile, sliceBudgetMs: options.sliceBudgetMs, + jobs: options.jobs, evalsAll: process.env.EVALS_ALL === '1', timeoutMs: timeoutOverride }); + console.log(`[test:paid] slice plan preview: tier=${manifest.tier} profile=${manifest.profile ?? 'full'} (${manifest.selectionReason})`); + for (const line of formatSlicePlan(manifest)) console.log(line); + return 0; + } + const { selected, excluded } = selectPaidTestFiles(discovered, options.tier); - const shards = planPaidShards(selected, { maxFilesPerShard: options.maxFilesPerShard }); + const shards = planPaidShards(expandCaseShards(selected, options.tier), { maxFilesPerShard: options.maxFilesPerShard }); // Parent-side diff selection (D9): skip whole shards whose mapped tests are // all unselected. Fail-open everywhere — the child's self-skip stays @@ -1862,7 +2211,7 @@ async function main(): Promise { const key = shards[index].join(' '); const note = skipReasons.has(key) ? ` [would skip: ${skipReasons.get(key)}]` : ''; const budget = resolvePaidShardBudget(shards[index], options.timeoutExplicit ? options.timeoutMs : undefined); - console.log(` shard ${index + 1}/${shards.length}: ${key} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'}${note}`); + console.log(` shard ${index + 1}/${shards.length}: ${key} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'} retries=${retriesForFiles(shards[index])}${note}`); } if (excluded.length > 0) { console.log(`\nExcluded (${excluded.length}):`); diff --git a/test/ci-native-evidence.test.ts b/test/ci-native-evidence.test.ts index 4cd903435..3d1b2ed3a 100644 --- a/test/ci-native-evidence.test.ts +++ b/test/ci-native-evidence.test.ts @@ -5,7 +5,7 @@ import * as path from 'node:path'; import { runPaidShard, shardSlug } from '../scripts/test-paid-shards'; const ROOT = path.resolve(import.meta.dir, '..'); -const workflows = ['evals.yml', 'evals-periodic.yml'].map(name => ({ +const workflows = ['evals.yml', 'evals-periodic.yml', 'evals-marathon.yml'].map(name => ({ name, value: Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows', name), 'utf8')) as any, })); @@ -23,7 +23,7 @@ function render(template: string, fields: Record): string { test('every direct CI paid executor binds a safe unique run/attempt/job/slice identity', () => { expect(executors.map(({ name, jobName }) => `${name}:${jobName}`)).toEqual([ - 'evals.yml:eval-slices', 'evals-periodic.yml:eval-slices', 'evals-periodic.yml:gate-census', + 'evals.yml:eval-slices', 'evals-periodic.yml:eval-slices', 'evals-periodic.yml:gate-census', 'evals-marathon.yml:eval-slices', ]); const ids = new Set(); for (const [workflowIndex, { job, step }] of executors.entries()) { @@ -32,9 +32,11 @@ test('every direct CI paid executor binds a safe unique run/attempt/job/slice id expect(env.EVALS_RUN_ID).toBeString(); for (const run of ['36302678692', '36302678693']) { for (const attempt of ['1', '2']) { - for (const slice of job.strategy.matrix.slice) { + // The planner sizes the matrix; cover more slices than any live plan. + expect(job.strategy.matrix.slice).toMatch(/^\$\{\{ fromJSON\(needs\.plan-slices\.outputs\.(?:[a-z]+_)?slices\) \}\}$/); + for (let slice = 1; slice <= 64; slice++) { const id = render(env.EVALS_RUN_ID, { - 'github.run_id': `${run}${workflowIndex === 0 ? '0' : '1'}`, + 'github.run_id': `${run}${workflowIndex}`, 'github.run_attempt': attempt, 'matrix.slice': String(slice), }); expect(id).toMatch(/^[A-Za-z0-9_-]+$/); diff --git a/test/cookie-validation-phases.test.ts b/test/cookie-validation-phases.test.ts index a4c8ce22a..53f0cd6dc 100644 --- a/test/cookie-validation-phases.test.ts +++ b/test/cookie-validation-phases.test.ts @@ -30,8 +30,10 @@ test('the actual CI cookie repair planner executes only eight dependent cases wi expect(manifest.evalsAll).toBe(false); expect(manifest.selection).toEqual({ e2e: ['browse-basic', 'browse-snapshot', 'qa-quick', 'qa-only-no-fix', 'design-review-detector-shim-dom', 'diagram-triplet', 'canary-workflow', 'benchmark-workflow'], judges: [] }); expect(manifest.entries.filter(entry => entry.status === 'planned').map(entry => entry.file).sort()).toEqual([ - 'test/skill-e2e-bws.test.ts', 'test/skill-e2e-deploy.test.ts', 'test/skill-e2e-design.test.ts', 'test/skill-e2e-diagram.test.ts', 'test/skill-e2e-qa-workflow.test.ts', + 'test/skill-e2e-bws.test.ts', 'test/skill-e2e-deploy.test.ts', 'test/skill-e2e-design.test.ts#design-review-detector-shim-dom', 'test/skill-e2e-diagram.test.ts', 'test/skill-e2e-qa-workflow.test.ts', ]); + // The case-sharded design file runs only its one selected cookie case. + expect(manifest.entries.filter(entry => entry.file.startsWith('test/skill-e2e-design.test.ts#') && entry.status === 'skipped-by-diff').length).toBeGreaterThan(0); }); test('the existing quality and behavior phases retain their complete separate shard census', () => { @@ -42,7 +44,10 @@ test('the existing quality and behavior phases retain their complete separate sh expect(quality.evalsAll).toBe(true); expect(behavior.evalsAll).toBe(true); expect(qualityFiles).toHaveLength(1); - expect(behaviorFiles).toHaveLength(45); + // 44 files (first-task-scaffold registers no gate case, so the gate lane + // skips it); the five case-sharded files contribute one shard per gate case. + expect(new Set(behaviorFiles.map(file => file.split('#')[0])).size).toBe(44); + expect(behaviorFiles).toHaveLength(62); expect(behaviorFiles).toEqual(expect.arrayContaining([ 'test/skill-e2e-qa-callers.test.ts', 'test/skill-e2e-qa-functional-fix.test.ts', diff --git a/test/e2e-shard-reuse.test.ts b/test/e2e-shard-reuse.test.ts new file mode 100644 index 000000000..a07b8f8e5 --- /dev/null +++ b/test/e2e-shard-reuse.test.ts @@ -0,0 +1,145 @@ +import { describe, expect, test } from 'bun:test'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { + e2eReuseEnvironment, e2eReuseLaneProblem, e2eShardIdentity, e2eShardInputFiles, prepareE2EShardReuse, + type E2EShardReuseRequest, +} from '../scripts/e2e-shard-reuse'; +import { buildRunManifest, fileCaseRegistration, runPaidShard, verifySliceResults, type SliceResult } from '../scripts/test-paid-shards'; + +const ROOT = path.resolve(import.meta.dir, '..'); +const FILE = 'test/skill-e2e-deploy.test.ts'; +const scratch = fs.mkdtempSync(path.join(os.tmpdir(), 'e2e-reuse-')); +const bin = path.join(scratch, 'bin'); +fs.mkdirSync(bin); +fs.writeFileSync(path.join(bin, 'claude'), '#!/bin/sh\necho "9.9.9 (Claude Code)"\n', { mode: 0o755 }); + +const laneEnv = (over: NodeJS.ProcessEnv = {}): NodeJS.ProcessEnv => ({ + PATH: `${bin}${path.delimiter}${process.env.PATH}`, HOME: scratch, + EVALS_TIER: 'gate', EVALS_PROFILE: 'pr', EVALS: '1', + EVALS_CACHE_DIR: path.join(scratch, 'cache'), EVALS_CACHE_REPOSITORY: 'garrytan/gstack', EVALS_CACHE_PR: '42', + EVALS_CACHE_RUNTIME_ID: 'a'.repeat(64), GITHUB_RUN_ID: '1001', GITHUB_RUN_ATTEMPT: '1', ANTHROPIC_API_KEY: 'sk-fixture', + GSTACK_CLAUDE_CLI_VERSION: '9.9.9 (Claude Code)', + ...over, +}); + +function request(over: Partial = {}): E2EShardReuseRequest { + const { registered, known } = fileCaseRegistration(FILE, fs.readFileSync(path.join(ROOT, FILE), 'utf8')); + return { root: ROOT, key: FILE, file: FILE, caseIds: ['setup-deploy-workflow'], registeredIds: registered, registrationKnown: known, + casePattern: '(?:^|\\s)(?:setup-deploy-workflow)$', expectedCases: 1, retries: 0, timeoutMs: 1_800_000, + withinShardConcurrency: 2, tier: 'gate', profile: 'pr', env: laneEnv(), ...over }; +} + +describe('E2E shard reuse eligibility', () => { + test('only the same-PR fast profile with an immutable runtime and default endpoint may reuse', () => { + expect(e2eReuseLaneProblem(laneEnv(), 'pr')).toBeNull(); + for (const [env, mode, problem] of [ + [laneEnv(), 'full-fallback', 'Only the fast PR profile'], + [laneEnv(), undefined, 'Only the fast PR profile'], + [laneEnv({ EVALS_CACHE_PR: '' }), 'pr', 'same-PR cache scope'], + [laneEnv({ EVALS_CACHE_RUNTIME_ID: 'latest' }), 'pr', 'immutable runtime'], + [laneEnv({ EVALS_FRESH: '1' }), 'pr', 'Fresh validation'], + [laneEnv({ EVALS_TIER: 'periodic' }), 'pr', 'Fresh validation'], + [laneEnv({ EVALS_CACHE_PURPOSE: 'periodic' }), 'pr', 'execute fresh'], + [laneEnv({ EVALS_CACHE_PURPOSE: 'marathon' }), 'pr', 'execute fresh'], + [laneEnv({ EVALS_CACHE_PURPOSE: 'release' }), 'pr', 'execute fresh'], + [laneEnv({ NODE_OPTIONS: '--require x' }), 'pr', 'Preload'], + [laneEnv({ ANTHROPIC_BASE_URL: 'https://proxy.example' }), 'pr', 'Custom model endpoint'], + ] as const) expect(e2eReuseLaneProblem(env, mode)).toContain(problem); + }); + + test('the identity binds the child environment except run-scoped transport, and never secret values', () => { + const env = e2eReuseEnvironment(laneEnv({ EVALS_RUN_ID: 'run-1', GSTACK_EVAL_DIR: '/tmp/x', EVALS_SELECTION_JSON: '{}', EVALS_MODEL: 'm', UNRELATED: 'x' })); + expect(env.ANTHROPIC_API_KEY).toBe('set'); + expect(env.EVALS_MODEL).toBe('m'); + for (const name of ['EVALS_RUN_ID', 'GSTACK_EVAL_DIR', 'EVALS_SELECTION_JSON', 'EVALS_CACHE_DIR', 'EVALS_CACHE_PR', 'UNRELATED', 'GITHUB_RUN_ID']) { + expect(env[name], name).toBeUndefined(); + } + expect(JSON.stringify(env)).not.toContain('sk-fixture'); + }); + + test('consumed files cover the test closure, every registered touchfile, the globals and the harness', () => { + const files = e2eShardInputFiles(request()); + for (const file of [FILE, 'test/helpers/e2e-helpers.ts', 'scripts/test-paid-shards.ts', 'scripts/e2e-shard-reuse.ts', + 'bun.lock', '.github/workflows/evals.yml', '.github/actions/register-gstack-skills/action.yml', '.github/docker/Dockerfile.ci', + 'setup-deploy/SKILL.md.tmpl', 'test/helpers/touchfiles-data.ts']) expect(files, file).toContain(file); + expect(files).not.toContain('package.json'); + expect(files.some(file => file.startsWith('node_modules/'))).toBe(true); + expect(() => e2eShardInputFiles({ ...request(), registeredIds: ['no-such-case'] })).not.toThrow(); + }); + + test('unknown or unprovable inputs fail closed', () => { + expect(e2eShardIdentity(request()).status).toBe('eligible'); + for (const [over, reason] of [ + [{ retries: 1 }, 'first attempt'], + [{ registrationKnown: false }, 'statically complete'], + [{ caseIds: [] }, 'exactly known'], + [{ expectedCases: 2 }, 'exactly known'], + [{ caseIds: ['not-registered'] }, 'exactly known'], + [{ file: 'test/skill-llm-eval.test.ts', key: 'test/skill-llm-eval.test.ts' }, 'audited E2E file'], + [{ env: laneEnv({ PATH: path.join(scratch, 'empty') }) }, 'Claude CLI version is unknown'], + ] as const) { + const result = e2eShardIdentity(request(over as Partial)); + expect(result.status, reason).toBe('ineligible'); + expect(result.status === 'ineligible' ? result.reason : '').toContain(reason); + } + }); + + test('any consumed parameter, pin or runtime change is a different identity', () => { + const key = (over: Partial) => { + const result = e2eShardIdentity(request(over)); + if (result.status !== 'eligible') throw new Error(result.reason); + return result.identity.key; + }; + const base = key({}); + expect(key({})).toBe(base); + expect(key({ env: laneEnv({ EVALS_RUN_ID: 'another-run', GITHUB_RUN_ID: '9' }) })).toBe(base); + for (const over of [{ timeoutMs: 1_000 }, { withinShardConcurrency: 1 }, { casePattern: 'x' }, + { env: laneEnv({ EVALS_MODEL: 'other' }) }, { env: laneEnv({ EVALS_CACHE_RUNTIME_ID: 'b'.repeat(64) }) }, + { env: laneEnv({ EVALS_CACHE_PR: '43' }) }] as Array>) expect(key(over)).not.toBe(base); + }); +}); + +describe('E2E shard reuse through the runner', () => { + test('a fresh first-attempt pass publishes; identical inputs then reuse without launching; changed inputs run', async () => { + const env = laneEnv({ EVALS_CACHE_DIR: path.join(scratch, 'roundtrip') }); + const first = prepareE2EShardReuse(request({ env }))!; + expect(first.lookup()).toBeNull(); + first.publish(); + const hit = prepareE2EShardReuse(request({ env }))!.lookup(); + expect(hit?.source.runId).toBe('1001/1'); + expect(prepareE2EShardReuse(request({ env: { ...env, EVALS_MODEL: 'changed' } }))!.lookup()).toBeNull(); + expect(prepareE2EShardReuse(request({ env: { ...env, EVALS_FRESH: '1' } }))).toBeNull(); + + const evalDir = path.join(scratch, 'evals'); + let launched = 0; + const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, evalDirBase: evalDir, env, log: () => {}, + expectedCaseIds: { [FILE]: ['setup-deploy-workflow'] }, + reuseFor: (files, childEnv) => prepareE2EShardReuse(request({ env: { ...childEnv } })), + commandFor: () => { launched++; return { command: process.execPath, args: ['-e', 'process.exit(1)'] }; } }); + expect(launched).toBe(0); + expect(outcome).toMatchObject({ status: 'passed', exitCode: 0, executedTests: 1, skippedTests: 0, reused: { runId: '1001/1' } }); + const recorded = JSON.parse(fs.readFileSync(path.join(evalDir, 'shards', 'skill-e2e-deploy', 'e2e-reused-skill-e2e-deploy.json'), 'utf8')); + expect(recorded.tests).toEqual([expect.objectContaining({ name: 'setup-deploy-workflow', passed: true, execution: 'reused' })]); + }); + + test('a failed shard never publishes a receipt', async () => { + let published = 0; + const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, env: laneEnv(), log: () => {}, + reuseFor: () => ({ lookup: () => null, publish: () => { published++; } }), + commandFor: () => ({ command: process.execPath, args: ['-e', 'process.exit(1)'] }) }); + expect(outcome.status).toBe('failed'); + expect(published).toBe(0); + }); + + test('the report accepts reused results only in the fast PR profile', () => { + const manifest = buildRunManifest({ tier: 'gate', sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } }); + const planned = manifest.entries.filter(entry => entry.status === 'planned'); + const reused = { inputKey: 'c'.repeat(64), runId: '1001/1', revision: 'd'.repeat(40), completedAt: 1 }; + const results: SliceResult[] = [{ version: 1, tier: 'gate', sliceIndex: 1, sliceCount: 1, outcomes: planned.map(entry => ({ + files: [entry.file], status: 'passed' as const, exitCode: 0, elapsedMs: 0, executedTests: 1, skippedTests: 0, + ...(entry.budget ? { budget: entry.budget } : {}), ...(entry.file === FILE ? { reused } : {}) })) }]; + expect(verifySliceResults(manifest, results).problems).toContain(`${FILE}: only the fast PR profile may reuse results; this lane executes fresh`); + }); +}); diff --git a/test/e2e-tier-alignment.test.ts b/test/e2e-tier-alignment.test.ts index e5f830792..1e4386ad4 100644 --- a/test/e2e-tier-alignment.test.ts +++ b/test/e2e-tier-alignment.test.ts @@ -184,8 +184,8 @@ describe('E2E tier alignment (touchfiles declaration vs test self-gate)', () => // Both self-gate shapes count: the raw predicate and the consolidated // helper (test/helpers/e2e-gate.ts documents this file as a consumer // that must recognize describeE2ETier/e2eTierEnabled). - const selfGated = /EVALS_TIER\s*===\s*['"](gate|periodic)['"]/.test(content) - || /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic)['"]/.test(content); + const selfGated = /EVALS_TIER\s*===\s*['"](gate|periodic|marathon)['"]/.test(content) + || /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic|marathon)['"]/.test(content); if (!usesNameSelection && selfGated) continue; // fail-open-safe standalone invisible.push( diff --git a/test/eng-finding-retry-budget.test.ts b/test/eng-finding-retry-budget.test.ts index 46b4ecaa8..b1b365eaa 100644 --- a/test/eng-finding-retry-budget.test.ts +++ b/test/eng-finding-retry-budget.test.ts @@ -1,5 +1,5 @@ import { expect, test } from 'bun:test'; -import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, OVERLAY_MAX_ACTIVE_SHARDS, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS } from '../scripts/test-paid-shards'; +import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards'; import { FINDING_RETRY_BUDGETS, ALL_TIERS, SHARD_RESERVE_MS } from './helpers/eval-budgets'; import fs from 'node:fs'; import os from 'node:os'; @@ -8,7 +8,8 @@ import path from 'node:path'; for (const budget of FINDING_RETRY_BUDGETS) { test(`${budget.file}: supervision preserves every existing attempt and retry`, () => { expect(budget.testMs).toBe(1_500_000); - expect(budget.retries).toBe(1); + // A 25-minute case is past RETRY_MAX_CASE_MS: a timed-out attempt is its verdict. + expect(budget.retries).toBe(0); expect(retriesForFiles([budget.file])).toBe(budget.retries); expect(budget.shardReserveMs).toBe(SHARD_RESERVE_MS); expect(budget.shardMs).toBe(budget.cases * budget.testMs * (budget.retries + 1) + budget.shardReserveMs); @@ -24,9 +25,8 @@ for (const budget of FINDING_RETRY_BUDGETS) { expect([...source.matchAll(/timeoutMs:\s*1_500_000\b/g)]).toHaveLength(budget.cases); } expect([...source.matchAll(/1_500_000\s*\/\* physical ceiling:/g)]).toHaveLength(budget.cases); - // Current periodic CI already supports this supervision wall. - const workflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any; - expect(budget.shardMs).toBeLessThan(workflow.jobs['eval-slices']['timeout-minutes'] * 60_000); + // The planned periodic CI job cap supports this supervision wall. + expect(budget.shardMs).toBeLessThan(livePlan().plan!.ciTimeoutMinutes * 60_000); }); test(`${budget.file}: own-shard allocation leaves ordinary and explicit limits intact`, () => { @@ -104,31 +104,36 @@ test('actual shard launcher honors the explicit saved planner limit without a pr } finally { fs.rmSync(dir, { recursive: true, force: true }); } }, 10000); -const periodicWorkflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any; -const periodicJob = periodicWorkflow.jobs['eval-slices']; -const periodicPlanStep = periodicWorkflow.jobs['plan-slices'].steps.find((step: any) => step.run?.includes('--tier periodic --emit-plan')); -const periodicSliceCount = Number(periodicPlanStep.run.match(/--slices\s+(\d+)/)?.[1]); -const periodicRunStep = periodicJob.steps.find((step: any) => step.run?.includes('--plan /tmp/paid-plan/manifest.json')); -const periodicWorkers = Number(periodicRunStep.env.EVALS_JOBS); -const livePlan = (discovered?: string[]) => buildRunManifest({ tier: 'periodic', sliceCount: periodicSliceCount, - evalsAll: true, env: { EVALS_ALL: '1' }, discovered }); +function periodicLane() { + const workflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any; + const job = workflow.jobs['eval-slices']; + const planStep = workflow.jobs['plan-slices'].steps.find((step: any) => step.run?.includes('--tier periodic --emit-plan')); + const planned = parseCliOptions(planStep.run.slice(planStep.run.indexOf('scripts/test-paid-shards.ts') + 'scripts/test-paid-shards.ts'.length).trim().split(/\s+/), {}); + const runStep = job.steps.find((step: any) => step.run?.includes('--plan /tmp/paid-plan/manifest.json')); + return { job, planStep, planned, workers: Number(runStep.env.EVALS_JOBS) }; +} +function livePlan(discovered?: string[]) { + const { planned } = periodicLane(); + return buildRunManifest({ tier: 'periodic', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs, + evalsAll: true, env: { EVALS_ALL: '1' }, discovered }); +} test('live periodic census fits the declared CI wall including setup', () => { + const { job, planStep, planned, workers } = periodicLane(); const m = livePlan(); - expect(periodicPlanStep.run).not.toContain('--autoplan-slice'); - expect(periodicJob.strategy.matrix.slice).toEqual(Array.from({ length: periodicSliceCount }, (_, index) => index + 1)); - expect(periodicWorkers).toBe(2); - const walls = Array.from({ length: periodicSliceCount }, (_, index) => { - const files = m.entries.filter(e => e.status === 'planned' && e.slice === index + 1).map(e => e.file); - const workers = files.some(isOverlayTestFile) ? Math.min(periodicWorkers, OVERLAY_MAX_ACTIVE_SHARDS) : periodicWorkers; - return paidShardWallUpperBoundMs(files, workers); - }); - expect(Math.max(...walls)).toBe(14_680_000); - expect(periodicJob['timeout-minutes']).toBe(360); - expect(periodicJob.strategy['max-parallel']).toBe(8); - expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(periodicJob['timeout-minutes'] * 60_000); - expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(70); - const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount); + expect(planStep.run).not.toContain('--autoplan-slice'); + expect(job.strategy.matrix.slice).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}'); + expect(job['timeout-minutes']).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}'); + expect(workers).toBe(2); + expect(planned.jobs).toBe(workers); + const walls = Array.from({ length: m.sliceCount }, (_, index) => sliceSupervisedWallMs(sliceExecutionOrder( + m.entries.filter(e => e.status === 'planned' && e.slice === index + 1)).map(e => e.file), workers)); + expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(m.plan!.ciTimeoutMinutes * 60_000); + expect(m.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360); + expect(m.sliceCount).toBeLessThanOrEqual(job.strategy['max-parallel']); + const plannedFiles = new Set(m.entries.filter(e => e.status === 'planned').map(e => shardFile(e.file))); + expect(plannedFiles).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected)); + const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === m.sliceCount); expect(overlays).toHaveLength(4); expect(overlays.every(e => isOverlayTestFile(e.file))).toBe(true); }); @@ -139,8 +144,8 @@ test('registered allocation is deterministic and preserves every discovered file expect(files).toContain('test/skill-e2e-ship-skip.test.ts'); const m = livePlan(files); expect(livePlan([...files].reverse())).toEqual(m); - expect(m.entries.map(e => e.file).sort()).toEqual([...files].sort()); - expect(new Set(m.entries.map(e => e.file)).size).toBe(files.length); + expect([...new Set(m.entries.map(e => shardFile(e.file)))].sort()).toEqual([...files].sort()); + expect(new Set(m.entries.map(e => e.file)).size).toBe(m.entries.length); }); test('ordinary-only manifests retain round-robin allocation', () => { @@ -171,17 +176,18 @@ test('single-slice manifest retains all registered files with one allocation', ( test('current detach supervision covers the live-census floor', () => { const floorFor = (tier: 'gate' | 'periodic') => { - const files = selectPaidTestFiles(collectPaidTestFiles(), tier).selected; + // Case-sharded files contribute one shard per case, exactly as the runner plans. + const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier); const excess = files.reduce((n, file) => n + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - DEFAULT_SHARD_TIMEOUT_MS), 0); return Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05); }; const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8')); const periodicTimeout = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]); const gateTimeout = Number(pkg.scripts['eval:bg:gate'].match(/--timeout\s+(\d+)/)[1]); - expect(floorFor('gate')).toBe(42_851); + expect(floorFor('gate')).toBe(26_597); expect(gateTimeout).toBe(49_320); expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate')); - expect(floorFor('periodic')).toBe(37_727); + expect(floorFor('periodic')).toBe(30_797); }); for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => { diff --git a/test/eval-detach-timeout-floor.test.ts b/test/eval-detach-timeout-floor.test.ts index fba4cba1c..b47ba816a 100644 --- a/test/eval-detach-timeout-floor.test.ts +++ b/test/eval-detach-timeout-floor.test.ts @@ -24,9 +24,10 @@ import { DEFAULT_JOBS, DEFAULT_SHARD_TIMEOUT_MS, resolvePaidShardBudget, + expandCaseShards, type PaidTier, } from '../scripts/test-paid-shards'; -import { FINDING_RETRY_BUDGETS } from './helpers/eval-budgets'; +import { FILE_RETRY_BUDGETS } from './helpers/eval-budgets'; const ROOT = path.resolve(import.meta.dir, '..'); // 5% margin over the theoretical bound: detach setup, lock wait, aggregation. @@ -57,7 +58,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => { ['periodic', 'eval:bg:periodic'], ] as Array<[PaidTier, string]>) { test(`${script} covers ordinary ${tier} waves plus registered excess x ${MARGIN}`, () => { - const files = selectPaidTestFiles(collectPaidTestFiles(), tier).selected; + const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier); expect(files.length).toBeGreaterThan(0); const floor = Math.ceil(worstCaseSeconds(files) * MARGIN); const configured = detachTimeoutSeconds(script); @@ -78,7 +79,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => { const pkg = JSON.parse(fs.readFileSync(path.join(ROOT, 'package.json'), 'utf8')); const jobs = Number(pkg.scripts['test:pr'].match(/EVALS_JOBS=\$\{EVALS_JOBS:-(\d+)\}/)?.[1]); expect(jobs).toBe(2); - const files = selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected; + const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected, 'gate'); const floor = Math.ceil(worstCaseSeconds(files, jobs) * MARGIN); expect(detachTimeoutSeconds('eval:bg:pr')).toBeGreaterThanOrEqual(floor); }); @@ -86,7 +87,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => { test('eval:bg:release covers both complete tiers and their existing margins', () => { const files = collectPaidTestFiles(); const floor = (['gate', 'periodic'] as const).reduce((sum, tier) => - sum + Math.ceil(worstCaseSeconds(selectPaidTestFiles(files, tier).selected) * MARGIN), 0); + sum + Math.ceil(worstCaseSeconds(expandCaseShards(selectPaidTestFiles(files, tier).selected, tier)) * MARGIN), 0); expect(detachTimeoutSeconds('eval:bg:release')).toBeGreaterThanOrEqual(floor); }); }); @@ -94,7 +95,9 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => { // One long job and one ordinary job can run side by side; the long job still // needs its whole wall, regardless of the number of ordinary workers. test('a heterogeneous pair rejects the old uniform-wall floor', () => { - const pair = [FINDING_RETRY_BUDGETS[0]!.file, 'test/skill-e2e-other.test.ts']; + // The longest registered wall (single-attempt finding files now fit the ordinary wall). + const longest = [...FILE_RETRY_BUDGETS].sort((a, b) => b.shardMs - a.shardMs)[0]!; + const pair = [longest.file, 'test/skill-e2e-other.test.ts']; const actualLongest = Math.max(...pair.map(file => resolvePaidShardBudget([file]).timeoutMs)) / 1000; expect(worstCaseSeconds(pair, 2)).toBe(actualLongest); expect(worstCaseSeconds(pair, 2)).toBeGreaterThan(DEFAULT_SHARD_TIMEOUT_MS / 1000); diff --git a/test/evals-workflow-wiring.test.ts b/test/evals-workflow-wiring.test.ts index 2e459fc17..3b34a2783 100644 --- a/test/evals-workflow-wiring.test.ts +++ b/test/evals-workflow-wiring.test.ts @@ -22,25 +22,53 @@ import { describe, test, expect } from 'bun:test'; import * as fs from 'fs'; import * as path from 'path'; -import { buildRunManifest, parseCliOptions, isOverlayTestFile, OVERLAY_MAX_ACTIVE_SHARDS, paidShardWallUpperBoundMs } from '../scripts/test-paid-shards'; +import { buildRunManifest, parseCliOptions, sliceExecutionOrder, sliceSupervisedWallMs, CI_SETUP_ALLOWANCE_MINUTES } from '../scripts/test-paid-shards'; const ROOT = path.join(import.meta.dir, '..'); const read = (rel: string) => fs.readFileSync(path.join(ROOT, rel), 'utf-8'); const evalsYml = read('.github/workflows/evals.yml'); const periodicYml = read('.github/workflows/evals-periodic.yml'); +const marathonYml = read('.github/workflows/evals-marathon.yml'); const registerAction = read('.github/actions/register-gstack-skills/action.yml'); -/** Slice count the planner emits (`--slices N`) in a workflow source. */ -function plannedSlices(source: string): number[] { - return [...source.matchAll(/--emit-plan\s+\S+\s+--slices\s+(\d+)/g)].map((m) => Number(m[1])); +/** Every planner site: its manifest path and budget (`--slice-budget S --jobs J`). */ +function plannerSites(source: string): Array<{ manifest: string; budgetSeconds: number; jobs: number }> { + return [...source.matchAll(/--emit-plan\s+(\S+)\s+--slice-budget\s+(\d+)\s+--jobs\s+(\d+)/g)] + .map((m) => ({ manifest: m[1]!, budgetSeconds: Number(m[2]), jobs: Number(m[3]) })); } -/** The executor matrix's slice list (`slice: [1, 2, ...]`). */ -function matrixSlices(source: string): number[][] { - return [...source.matchAll(/^\s+slice: \[([\d,\s]+)\]\s*$/gm)].map((m) => - m[1].split(',').map((n) => Number(n.trim())), - ); +type Step = { id?: string; name?: string; run?: string; env?: Record; with?: Record }; +type Job = { needs?: string[]; env?: Record; outputs?: Record; 'timeout-minutes': string | number; + strategy?: { 'max-parallel': number; matrix: { slice: string } }; steps: Step[] }; + +/** + * An executor's matrix and timeout must come from the planner step that wrote + * the manifest it downloads: `slices` from `[range(1; .sliceCount + 1)]` and + * `timeout-minutes` from `.plan.ciTimeoutMinutes`, never hand-written numbers. + */ +function expectPlannedExecutor(source: string, executorName: string, prefix: string) { + const workflow = Bun.YAML.parse(source) as { jobs: Record }; + const planner = workflow.jobs['plan-slices']!; + const executor = workflow.jobs[executorName]!; + expect(executor.needs).toContain('plan-slices'); + expect(executor.strategy!.matrix.slice).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}slices) }}`); + expect(executor['timeout-minutes']).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}timeout_minutes) }}`); + const [stepId] = /^\$\{\{ steps\.([\w-]+)\.outputs\.slices \}\}$/.exec(planner.outputs![`${prefix}slices`]!)!.slice(1); + expect(planner.outputs![`${prefix}timeout_minutes`]).toBe(`\${{ steps.${stepId}.outputs.timeout_minutes }}`); + const matrixStep = planner.steps.find(step => step.id === stepId)!; + const manifest = /jq -c '\[range\(1; \.sliceCount \+ 1\)\]' (\S+)\)/.exec(matrixStep.run!)![1]!; + expect(matrixStep.run).toContain(`jq -e '.plan.ciTimeoutMinutes' ${manifest})`); + const emit = planner.steps.filter(step => step.run?.includes(`--emit-plan ${manifest} `)); + expect(emit).toHaveLength(1); + const execute = executor.steps.filter(step => step.run?.includes('--plan ')); + expect(execute).toHaveLength(1); + expect(execute[0]!.run).toContain(`--plan ${manifest} --slice \${{ matrix.slice }}`); + expect(executor.steps.some(step => step.with?.path === manifest.replace(/\/manifest\.json$/, ''))).toBe(true); + // The planner packs for exactly the executor's worker count. + const site = plannerSites(emit[0]!.run!)[0]!; + expect(execute[0]!.env?.EVALS_JOBS).toBe(String(site.jobs)); + return { site, emit: emit[0]!, execute: execute[0]!, executor, planner }; } describe('evals.yml sliced-lane wiring (post-matrix)', () => { @@ -67,13 +95,12 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => { expect(evalsYml).toMatch(/EVALS_TIER=gate bun --no-install run scripts\/test-paid-shards\.ts --tier gate --report /); }); - test('executor matrix slice list matches the planner --slices count', () => { - const planned = plannedSlices(evalsYml); - const matrices = matrixSlices(evalsYml); - expect(planned, 'expected exactly one --emit-plan site in evals.yml').toHaveLength(1); - expect(matrices, 'expected exactly one slice matrix in evals.yml').toHaveLength(1); - const n = planned[0]; - expect(matrices[0]).toEqual(Array.from({ length: n }, (_, i) => i + 1)); + test('executor matrix and timeout come from the one budget planner', () => { + expect(plannerSites(evalsYml), 'expected exactly one --emit-plan site in evals.yml').toHaveLength(1); + const { site } = expectPlannedExecutor(evalsYml, 'eval-slices', ''); + expect(site).toEqual({ manifest: '/tmp/paid-plan/manifest.json', budgetSeconds: 540, jobs: 2 }); + // The validation-phase planner writes the same manifest with the same budget. + expect(evalsYml).toContain('sliceBudgetMs: 540000, jobs: 2'); }); test('reconcile exit is captured via PIPESTATUS, never $? after a pipe', () => { @@ -81,7 +108,7 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => { // `$?` after `... | tee` is tee's exit — always 0. That made the // fail-closed reconcile gate silently fail-open (ship review army, // 2026-08-31). Both lanes must read PIPESTATUS[0]. - for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) { + for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) { const reconcileBlocks = [...source.matchAll(/--report[^\n]*\| tee[^\n]*\n([\s\S]{0,400}?)GITHUB_OUTPUT/g)]; expect(reconcileBlocks.length, `${name}: expected a tee'd reconcile step`).toBeGreaterThanOrEqual(1); for (const block of reconcileBlocks) { @@ -124,78 +151,89 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => { }); describe('evals-periodic.yml sliced-lane wiring', () => { - test('the CI job cap covers the live periodic slice census plus setup', () => { - type Env = Record; - const workflow = Bun.YAML.parse(periodicYml) as { - env?: Env; - jobs: Record; - }>; - }; - const planner = workflow.jobs['plan-slices']; - const executor = workflow.jobs['eval-slices']; - const plannerSteps = planner.steps.filter(step => step.run?.includes('EVALS_TIER=periodic ') && step.run.includes('--emit-plan ')); - const executorSteps = executor.steps.filter(step => step.run?.includes('--plan ')); - expect(plannerSteps).toHaveLength(1); - expect(executorSteps).toHaveLength(1); - const cliArgs = (run: string) => { - const command = /\bbun(?: --no-install)? run scripts\/test-paid-shards\.ts /.exec(run); - expect(command).not.toBeNull(); - return run.slice(command!.index + command![0].length) - .replace(/\$\{\{\s*matrix\.slice\s*\}\}/g, '1').trim().split(/\s+/); - }; - const plannerEnv = { ...workflow.env, ...planner.env, ...plannerSteps[0].env }; - const plannerOptions = parseCliOptions(cliArgs(plannerSteps[0].run!), plannerEnv); - const executorOptions = parseCliOptions(cliArgs(executorSteps[0].run!), { - ...workflow.env, ...executor.env, ...executorSteps[0].env, - }); - expect(plannerEnv.EVALS_ALL).toBe('1'); - expect(plannerOptions.tier).toBe('periodic'); - expect(executorOptions.tier).toBe('periodic'); - const slices = executor.strategy!.matrix.slice; - expect(slices).toEqual(Array.from({ length: plannerOptions.slices }, (_, i) => i + 1)); - const manifest = buildRunManifest({ - tier: plannerOptions.tier, sliceCount: plannerOptions.slices, - evalsAll: true, env: plannerEnv, rootDir: ROOT, - }); - // Resolve the same per-file walls and overlay admission limit as execution. - const explicitWall = executorOptions.timeoutExplicit ? executorOptions.timeoutMs : undefined; - const setupAllowanceMinutes = 20; - const allowances = slices.map(slice => { - const files = manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice).map(entry => entry.file); - const normal = files.filter(file => !isOverlayTestFile(file)); - const overlay = files.filter(isOverlayTestFile); - const bound = (group: string[], jobs: number) => paidShardWallUpperBoundMs(group, jobs, explicitWall); - return (bound(normal, executorOptions.jobs) + bound(overlay, Math.min(executorOptions.jobs, OVERLAY_MAX_ACTIVE_SHARDS))) / 60_000; - }); - expect(Math.max(...allowances)).toBeGreaterThan(0); - const requiredMinutes = Math.max(...allowances) + setupAllowanceMinutes; - expect(executor['timeout-minutes'], - `periodic slice allowances ${allowances.join(', ')} minutes + ${setupAllowanceMinutes} minutes setup require ${requiredMinutes} CI minutes`, - ).toBeGreaterThanOrEqual(requiredMinutes); - }); + const lanes = [ + { source: periodicYml, name: 'evals-periodic.yml', executor: 'eval-slices', prefix: 'periodic_', tier: 'periodic' }, + { source: periodicYml, name: 'evals-periodic.yml', executor: 'gate-census', prefix: 'gate_', tier: 'gate' }, + { source: evalsYml, name: 'evals.yml', executor: 'eval-slices', prefix: '', tier: 'gate' }, + { source: marathonYml, name: 'evals-marathon.yml', executor: 'eval-slices', prefix: '', tier: 'marathon' }, + ] as const; - test('planner/executor/report tier=periodic and slice counts agree', () => { + for (const lane of lanes) { + test(`${lane.name}:${lane.executor} — the planned CI job cap covers every slice's supervised wall plus setup, and every slice starts at once`, () => { + const { emit, execute, executor } = expectPlannedExecutor(lane.source, lane.executor, lane.prefix); + const cliArgs = (run: string) => { + const command = /\bbun(?: --no-install)? run scripts\/test-paid-shards\.ts /.exec(run); + expect(command).not.toBeNull(); + return run.slice(command!.index + command![0].length) + .replace(/\$\{\{\s*matrix\.slice\s*\}\}/g, '1').trim().split(/\s+/); + }; + const workflow = Bun.YAML.parse(lane.source) as { env?: Record }; + // The complete census (EVALS_ALL) is the largest plan any event can produce. + const plannerEnv = { ...workflow.env, ...emit.env, EVALS_ALL: '1', EVALS_PROFILE: 'full' }; + const planned = parseCliOptions(cliArgs(emit.run!), plannerEnv); + const active = parseCliOptions(cliArgs(execute.run!), { ...workflow.env, ...executor.env, ...execute.env, EVALS_PROFILE: 'full' }); + expect(planned.tier).toBe(lane.tier); + expect(active.tier).toBe(lane.tier); + expect(active.jobs).toBe(planned.jobs); + const manifest = buildRunManifest({ tier: planned.tier, profile: 'full', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs, + evalsAll: true, env: plannerEnv, rootDir: ROOT, skipJudges: planned.skipJudges }); + const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder( + manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === i + 1)).map(entry => entry.file), planned.jobs)); + const requiredMinutes = Math.ceil(Math.max(0, ...walls) / 60_000) + CI_SETUP_ALLOWANCE_MINUTES; + expect(CI_SETUP_ALLOWANCE_MINUTES).toBe(20); + expect(manifest.plan!.ciTimeoutMinutes, `slice walls ${walls.join(', ')}ms`).toBe(requiredMinutes); + // GitHub-hosted-style job ceiling: a plan past it must be split, not truncated. + expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360); + expect(manifest.sliceCount, `${lane.name}:${lane.executor} plans more slices than max-parallel starts at once`) + .toBeLessThanOrEqual(executor.strategy!['max-parallel']); + }); + } + + test('planner/executor/report tier=periodic agree and plan with the ~9-minute budget', () => { expect(periodicYml).toMatch(/EVALS_TIER=periodic bun --no-install run scripts\/test-paid-shards\.ts --tier periodic --emit-plan/); expect(periodicYml).toMatch(/EVALS_TIER=periodic bun run scripts\/test-paid-shards\.ts --tier periodic --plan .* --slice /); expect(periodicYml).toMatch(/EVALS_TIER=periodic bun --no-install run scripts\/test-paid-shards\.ts --tier periodic --report /); - const planned = plannedSlices(periodicYml); - const matrices = matrixSlices(periodicYml); // Periodic work and the full gate census have distinct immutable plans. - expect(planned).toHaveLength(2); - expect(matrices).toHaveLength(2); - for (const [index, count] of planned.entries()) { - expect(matrices[index]).toEqual(Array.from({ length: count }, (_, i) => i + 1)); + expect(plannerSites(periodicYml)).toEqual([ + { manifest: '/tmp/paid-plan/manifest.json', budgetSeconds: 540, jobs: 2 }, + { manifest: '/tmp/gate-census-plan/manifest.json', budgetSeconds: 540, jobs: 2 }, + ]); + }); +}); + +describe('evals-marathon.yml non-blocking lane', () => { + const workflow = Bun.YAML.parse(marathonYml) as { on: Record; env: Record; jobs: Record }; + + test('runs weekly and on dispatch, always fresh, with its own fail-closed report and tracking issue', () => { + expect(Object.keys(workflow.on).sort()).toEqual(['schedule', 'workflow_dispatch']); + expect(workflow.env).toMatchObject({ EVALS_PROFILE: 'full', EVALS_FRESH: '1', EVALS_CACHE_PURPOSE: 'marathon' }); + expect(marathonYml).not.toContain('actions/cache'); + expect(marathonYml).toMatch(/EVALS_TIER=marathon bun --no-install run scripts\/test-paid-shards\.ts --tier marathon --emit-plan \/tmp\/marathon-plan\/manifest\.json --slice-budget 1 --jobs 1/); + expect(marathonYml).toMatch(/EVALS_TIER=marathon bun run scripts\/test-paid-shards\.ts --tier marathon --plan .* --slice /); + const report = workflow.jobs.report!; + expect(report.needs).toEqual(['plan-slices', 'eval-slices']); + const reconcile = report.steps.find(step => step.id === 'reconcile')!; + expect(reconcile.run).toContain('EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --report /tmp/marathon-report'); + const guards = report.steps.filter(step => /Upsert tracking|Fail the workflow/.test(step.name ?? '')); + expect(guards).toHaveLength(2); + for (const step of guards) { + expect((step as { if?: string }).if).toContain("steps.reconcile.outputs.exit != '0'"); + expect((step as { if?: string }).if).toContain("needs.eval-slices.result != 'success'"); + } + expect(marathonYml).toContain('Weekly marathon evals: red lane needs triage'); + }); + + test('the blocking lanes never plan or execute the marathon tier', () => { + for (const source of [evalsYml, periodicYml]) { + expect(source).not.toContain('--tier marathon'); + expect(source).not.toContain('EVALS_TIER=marathon'); } }); }); -describe('shared setup composites (both surviving lanes)', () => { - test('both lanes register skills through the shared composite', () => { - for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) { +describe('shared setup composites (every paid lane)', () => { + test('every lane registers skills through the shared composite', () => { + for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) { expect(source, `${name} must use the register-gstack-skills composite`) .toContain('uses: ./.github/actions/register-gstack-skills'); // No inline re-implementation creeping back beside the composite. @@ -218,7 +256,7 @@ describe('shared setup composites (both surviving lanes)', () => { for (const action of ['seed-claude-config', 'restore-deps', 'fix-bun-temp']) { expect(fs.existsSync(path.join(ROOT, '.github', 'actions', action, 'action.yml')), `missing composite: ${action}`).toBe(true); } - for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) { + for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) { expect(source, `${name} must use seed-claude-config`).toContain('uses: ./.github/actions/seed-claude-config'); expect(source, `${name} must use restore-deps`).toContain('uses: ./.github/actions/restore-deps'); expect(source, `${name} must use fix-bun-temp`).toContain('uses: ./.github/actions/fix-bun-temp'); diff --git a/test/helpers/eval-budgets.ts b/test/helpers/eval-budgets.ts index dd13b9783..06bdfd664 100644 --- a/test/helpers/eval-budgets.ts +++ b/test/helpers/eval-budgets.ts @@ -46,70 +46,133 @@ export const ALL_TIERS = { /** Supervision reserve added to every registered whole-file wall. */ export const SHARD_RESERVE_MS = 2 * 60_000; -/** Whole-file supervision must cover each existing attempt and its retry. - * These fixtures already allow 25 minutes per case; the old 30-minute - * wall could kill a second attempt after five minutes. No case budget grows. +/** + * Retry policy (approved 2026-09-29): a timed-out attempt is a verdict. Bun's + * --retry reruns a failed case after it may have spent its whole budget, so an + * automatic retry is kept only where one more attempt is short: every case of + * the file has a per-attempt budget of at most RETRY_MAX_CASE_MS, the CAPTURE + * tier plus its recording grace. Those failures are fast flake classes (API + * blips, tool hiccups) and a retry costs at most one more short attempt. Files + * with any longer case run once. Per-case budgets never change with this rule. + */ +export const RETRY_MAX_CASE_MS = CAPTURE_MS + 15_000; + +export function retriesWithinCaseCap(caseMs: number, configuredRetries: number): number { + return caseMs <= RETRY_MAX_CASE_MS ? configuredRetries : 0; +} + +/** + * Unregistered paid files that keep one automatic retry: every case budget is + * JUDGE or CAPTURE tier (test/paid-retry-supervision.test.ts scans each source). + * Registered rows below derive retries from their declared caseMs; every other + * paid file runs once. + */ +export const SHORT_CASE_RETRY_FILES: readonly string[] = [ + 'test/codex-e2e-sol-scope.test.ts', + 'test/llm-judge-recommendation.test.ts', + 'test/skill-e2e-ask-user-question-format-compliance.test.ts', + 'test/skill-e2e-benchmark-providers.test.ts', + 'test/skill-e2e-bws.test.ts', + 'test/skill-e2e-context-skills.test.ts', + 'test/skill-e2e-coverage-audit.test.ts', + 'test/skill-e2e-diagram.test.ts', + 'test/skill-e2e-first-task-scaffold.test.ts', + 'test/skill-e2e-gbrain-roundtrip-local.test.ts', + 'test/skill-e2e-hermetic-canary.test.ts', + 'test/skill-e2e-investigate-owned-completion.test.ts', + 'test/skill-e2e-investigate-owned-termination.test.ts', + 'test/skill-e2e-learnings.test.ts', + 'test/skill-e2e-plan-tune.test.ts', + 'test/skill-e2e-qa-functional-fix.test.ts', + 'test/skill-e2e-qa-functional.test.ts', + 'test/skill-e2e-review-army.test.ts', + 'test/skill-e2e-review.test.ts', + 'test/skill-e2e-session-intelligence.test.ts', + 'test/skill-e2e-setup-gbrain-bad-token.test.ts', + 'test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts', + 'test/skill-e2e-setup-gbrain-remote.test.ts', + 'test/skill-e2e-ship-hook-consent.test.ts', + 'test/skill-e2e-ship-hook-refresh.test.ts', + 'test/skill-e2e-ship-skip.test.ts', + 'test/skill-e2e-sync-gbrain-readiness.test.ts', + 'test/skill-e2e-third-party-actions.test.ts', + 'test/skill-e2e-triage.test.ts', + 'test/skill-routing-e2e.test.ts', +]; + +/** Whole-file supervision covers every attempt the retry policy allows. + * These fixtures allow 25 minutes per case, so they run once. * Reserve the sequential upper bound even when Bun runs sibling cases together. */ export const FINDING_RETRY_BUDGETS = [ { file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 }, { file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 }, -].map(({ file, cases }) => ({ - file, cases, - id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`, - testMs: 1_500_000, - retries: 1, - shardReserveMs: SHARD_RESERVE_MS, - shardMs: cases * 1_500_000 * 2 + SHARD_RESERVE_MS, -})); +].map(({ file, cases }) => { + const retries = retriesWithinCaseCap(1_500_000, 1); + return { + file, cases, + id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`, + testMs: 1_500_000, + caseMs: 1_500_000, + retries, + shardReserveMs: SHARD_RESERVE_MS, + shardMs: cases * 1_500_000 * (retries + 1) + SHARD_RESERVE_MS, + }; +}); -/** Three existing captures and one configured retry; only supervision grows. */ +/** Three existing captures in one 16-minute case, so the file runs once. */ export const AUQ_CONSISTENCY_RETRY_BUDGET = { file: 'test/skill-e2e-auq-consistency.test.ts', id: 'auq-consistency-existing-retry-v1', cases: 1, testMs: 3 * CAPTURE_MS + 60_000, - retries: 1, + caseMs: 3 * CAPTURE_MS + 60_000, + retries: retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1), shardReserveMs: SHARD_RESERVE_MS, - shardMs: (3 * CAPTURE_MS + 60_000) * 2 + SHARD_RESERVE_MS, + shardMs: (3 * CAPTURE_MS + 60_000) * (retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1) + 1) + SHARD_RESERVE_MS, } as const; /** These fixtures have a fixed case count in every supported tier. */ export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET]; -/** Whole-file walls cover all existing cases and retries, even if Bun runs them - * sequentially. Mixed-tier files reserve their larger complete tier, never a - * currently selected subset. These rows add no case-count or model-work policy. - * The 10-second terms preserve the existing Codex/recording finalization grace. +/** Whole-file walls cover all existing cases and every allowed attempt, even if + * Bun runs them sequentially. Mixed-tier files reserve their larger complete + * tier, never a currently selected subset. caseMs is the longest single case + * budget, which decides the retry (RETRY_MAX_CASE_MS). These rows add no + * case-count or model-work policy. The 10-second terms preserve the existing + * Codex/recording finalization grace. */ export const FILE_RETRY_BUDGETS = [ ...STRICT_RETRY_CASE_BUDGETS, ...[ - { file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), retries: 1 }, - { file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, retries: 1 }, - { file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 5 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, retries: 1 }, + { file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000, configuredRetries: 1 }, + { file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 5 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, // Seventeen workflow judges include their 10s recording grace; the other // seven judges retain 120s. Supervise all 24 and the existing one retry. - { file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, retries: 1 }, - { file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 }, - { file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 }, - { file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, retries: 1 }, - { file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 }, - { file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 }, - { file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 }, - { file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 }, - { file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, retries: 2 }, - { file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, retries: 1 }, - { file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, retries: 1 }, - { file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 }, + { file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000, configuredRetries: 1 }, + { file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 }, + { file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 2 }, + { file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 }, + { file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 }, // Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s. - { file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), retries: 1 }, - ].map(({ file, attemptMs, retries }) => ({ - file, attemptMs, retries, - id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`, - shardReserveMs: SHARD_RESERVE_MS, - shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS, - })), + { file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS, configuredRetries: 1 }, + ].map(({ file, attemptMs, caseMs, configuredRetries }) => { + const retries = retriesWithinCaseCap(caseMs, configuredRetries); + return { + file, attemptMs, caseMs, retries, + id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`, + shardReserveMs: SHARD_RESERVE_MS, + shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS, + }; + }), ]; /** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */ diff --git a/test/paid-overlay-scheduling.test.ts b/test/paid-overlay-scheduling.test.ts index ca683b857..701eb5c07 100644 --- a/test/paid-overlay-scheduling.test.ts +++ b/test/paid-overlay-scheduling.test.ts @@ -22,17 +22,18 @@ const fakeEnv = { describe('overlay file policy', () => { test('grouped planning isolates every overlay and preserves ordinary retries', () => { - const workflow = 'test/skill-e2e-workflow.test.ts'; - const files = [...overlayFiles, normalFile, workflow]; + // Two short-case files keep their one retry (timeout-is-a-verdict rule). + const workflow = 'test/skill-e2e-review.test.ts'; + const files = [...overlayFiles, 'test/skill-e2e-triage.test.ts', workflow]; for (const maxFilesPerShard of [2, 3, 10]) { const shards = planPaidShards(files, { maxFilesPerShard }); expect(shards.flat().sort()).toEqual([...files].sort()); for (const file of overlayFiles) expect(shards).toContainEqual([file]); const workflowShard = shards.find(shard => shard.includes(workflow))!; expect(workflowShard.some(isOverlayTestFile)).toBe(false); - expect(retriesForFiles(workflowShard)).toBe(2); + expect(retriesForFiles(workflowShard)).toBe(1); const args = buildPaidShardArgs(workflowShard, resolvePaidShardTimeoutMs(workflowShard), 2, retriesForFiles(workflowShard)); - expect(args[args.indexOf('--retry') + 1]).toBe('2'); + expect(args[args.indexOf('--retry') + 1]).toBe('1'); expect(planPaidShards(files.map(file => file.replaceAll('/', '\\')), { maxFilesPerShard })).toEqual(shards); } }); @@ -65,8 +66,10 @@ describe('overlay file policy', () => { for (const file of [normalFile, 'test/skill-e2e-overlay-harness.test.ts', 'test/model-overlays.test.ts']) { expect(isOverlayTestFile(file)).toBe(false); expect(resolvePaidShardTimeoutMs([file])).toBe(DEFAULT_SHARD_TIMEOUT_MS); - expect(retriesForFiles([file])).toBe(1); + // Not overlays; unlisted files run once because their case budget is unknown. + expect(retriesForFiles([file])).toBe(0); } + expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1); expect(resolvePaidShardTimeoutMs([normalFile], 1234)).toBe(1234); expect(resolvePaidShardTimeoutMs([overlayFiles[0]], 1_900_000)).toBe(1_900_000); expect(() => resolvePaidShardTimeoutMs([overlayFiles[0]], 1_800_000)).toThrow('explicit wall'); @@ -157,8 +160,8 @@ describe('overlay manifest affinity and CI capacity', () => { const workflow = Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows/evals-periodic.yml'), 'utf8')) as { jobs: Record; - strategy: { matrix: { slice: number[] } }; - 'timeout-minutes': number; + strategy: { matrix: { slice: string } }; + 'timeout-minutes': string; }>; }; const job = workflow.jobs['eval-slices']; @@ -166,13 +169,20 @@ describe('overlay manifest affinity and CI capacity', () => { const jobs = parseCliOptions([], step.env).jobs; expect(jobs).toBe(2); expect(parseCliOptions([], step.env).withinShardConcurrency).toBe(2); - expect(job.strategy.matrix.slice).toEqual([1, 2, 3, 4, 5, 6, 7]); + expect(job.strategy.matrix.slice).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}'); + expect(job['timeout-minutes']).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}'); const normalMinutes = Math.ceil(18 / jobs) * resolvePaidShardTimeoutMs([normalFiles[0]]) / 60_000; const overlayMinutes = Math.ceil(overlayFiles.length / OVERLAY_MAX_ACTIVE_SHARDS) * Math.max(...overlayFiles.map(file => resolvePaidShardTimeoutMs([file]))) / 60_000; expect(normalMinutes).toBe(270); expect(overlayMinutes).toBe(122); - expect(job['timeout-minutes']).toBeGreaterThanOrEqual(Math.max(normalMinutes, overlayMinutes) + 20); + // The CI budget plan (what the workflow runs) keeps the four overlays + // in one final one-at-a-time slice and its job cap covers them. + const budget = buildRunManifest({ ...opts, sliceCount: undefined, sliceBudgetMs: 540_000, jobs }); + const lastSlice = budget.entries.filter(e => e.slice === budget.sliceCount).map(e => e.file).sort(); + expect(lastSlice).toEqual([...overlayFiles].sort()); + expect(budget.plan!.ciTimeoutMinutes).toBeGreaterThanOrEqual(overlayMinutes + 20); + expect(budget.plan!.ciTimeoutMinutes).toBe(Math.ceil(overlayMinutes) + 20); // Gate selection keeps its original periodic exclusion and all six // ordinary slices; reservation does not spend an empty slot in gate. diff --git a/test/paid-pr-profile.test.ts b/test/paid-pr-profile.test.ts index 8130ee419..34609e54f 100644 --- a/test/paid-pr-profile.test.ts +++ b/test/paid-pr-profile.test.ts @@ -150,7 +150,8 @@ describe('PR profile paid-runner integration', () => { expect(() => parseRunManifest(JSON.stringify(injected))).toThrow('outside its PR case selection'); for (const action of ['remove', 'skip', 'duplicate'] as const) { const missing = structuredClone(manifest); - const file = 'test/skill-e2e-plan.test.ts'; + // plan.test is case-sharded: its PR case runs as `#`. + const file = manifest.entries.find(entry => entry.status === 'planned' && entry.file.startsWith('test/skill-e2e-plan.test.ts#'))!.file; if (action === 'remove') missing.entries = missing.entries.filter(entry => entry.file !== file); if (action === 'skip') missing.entries.find(entry => entry.file === file)!.status = 'skipped-by-diff'; if (action === 'duplicate') missing.entries.push({ ...missing.entries.find(entry => entry.file === file)! }); diff --git a/test/paid-retry-supervision.test.ts b/test/paid-retry-supervision.test.ts index 40711e083..ba231e79a 100644 --- a/test/paid-retry-supervision.test.ts +++ b/test/paid-retry-supervision.test.ts @@ -4,34 +4,69 @@ import { join } from 'node:path'; import { buildPaidShardArgs, buildRunManifest, parseRunManifest, planPaidShards, DEFAULT_JOBS, parseCliOptions, paidShardWallUpperBoundMs, resolvePaidShardBudget, retriesForFiles, verifySliceResults, collectPaidTestFiles, selectPaidTestFiles, + shardFile, sliceExecutionOrder, sliceSupervisedWallMs, CASE_SHARDED_FILES, } from '../scripts/test-paid-shards'; import { - ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, + ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, RETRY_MAX_CASE_MS, SHORT_CASE_RETRY_FILES, FINDING_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS, } from './helpers/eval-budgets'; +import { E2E_TOUCHFILES } from './helpers/touchfiles'; + const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8'); const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file)); +// Walls cover every attempt the retry rule allows: files with a case budget +// past RETRY_MAX_CASE_MS run once (a timed-out attempt is a verdict). const expectedWalls = { 'test/skill-e2e-qa-callers.test.ts': 3_270_000, - 'test/skill-e2e-shared-libs-paths.test.ts': 3_720_000, - 'test/skill-e2e-ship-docsync.test.ts': 10_920_000, + 'test/skill-e2e-shared-libs-paths.test.ts': 1_920_000, + 'test/skill-e2e-ship-docsync.test.ts': 5_520_000, 'test/skill-llm-eval.test.ts': 6_220_000, - 'test/skill-e2e-auq-consistency.test.ts': 2_040_000, + 'test/skill-e2e-auq-consistency.test.ts': 1_080_000, 'test/skill-e2e-auq-matrix.test.ts': 3_720_000, 'test/skill-e2e-plan-format.test.ts': 2_600_000, - 'test/skill-e2e-auto-decide-preserved.test.ts': 1_920_000, - 'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_920_000, - 'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_920_000, - 'test/skill-e2e-plan-design-finding-floor.test.ts': 1_920_000, - 'test/skill-e2e-plan-devex-finding-floor.test.ts': 1_920_000, - 'test/skill-e2e-plan-mode-no-op.test.ts': 9_120_000, - 'test/skill-e2e-plan-ceo-mode-routing.test.ts': 2_520_000, - 'test/skill-e2e-plan-eng-plan-mode.test.ts': 2_520_000, + 'test/skill-e2e-auto-decide-preserved.test.ts': 1_020_000, + 'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_020_000, + 'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_020_000, + 'test/skill-e2e-plan-design-finding-floor.test.ts': 1_020_000, + 'test/skill-e2e-plan-devex-finding-floor.test.ts': 1_020_000, + 'test/skill-e2e-plan-mode-no-op.test.ts': 3_120_000, + 'test/skill-e2e-plan-ceo-mode-routing.test.ts': 1_320_000, + 'test/skill-e2e-plan-eng-plan-mode.test.ts': 1_320_000, 'test/skill-e2e-plan-prosons.test.ts': 2_600_000, - 'test/skill-e2e-plan.test.ts': 7_320_000, + 'test/skill-e2e-plan.test.ts': 3_720_000, }; +test('retry rule: only files whose every case is CAPTURE tier or shorter retry; longer cases run once', () => { + expect(RETRY_MAX_CASE_MS).toBe(ALL_TIERS.CAPTURE_MS + 15_000); + for (const row of FILE_RETRY_BUDGETS) { + expect(row.retries, row.file).toBe(row.caseMs <= RETRY_MAX_CASE_MS ? (row.file.endsWith('plan-mode-no-op.test.ts') ? 2 : 1) : 0); + expect(retriesForFiles([row.file])).toBe(row.retries); + } + expect(FILE_RETRY_BUDGETS.filter(row => row.retries > 0).map(row => row.file).sort()).toEqual([ + 'test/skill-e2e-auq-matrix.test.ts', 'test/skill-e2e-plan-format.test.ts', 'test/skill-e2e-plan-prosons.test.ts', + 'test/skill-e2e-qa-callers.test.ts', 'test/skill-llm-eval.test.ts', + ]); + const paid = collectPaidTestFiles(); + for (const file of SHORT_CASE_RETRY_FILES) { + expect(paid, `stale SHORT_CASE_RETRY_FILES entry: ${file}`).toContain(file); + expect(FILE_RETRY_BUDGETS.some(row => row.file === file)).toBe(false); + const source = read(file); + // Declared short budgets only: a JUDGE/CAPTURE tier or a literal at most the + // cap, no longer tier and no ms literal past the cap. + const literals = [...source.matchAll(/(? Number(match[1]!.replace(/_/g, ''))); + expect(/\b(?:JUDGE_MS|CAPTURE_MS)\b/.test(source) || literals.some(ms => ms >= 60_000 && ms <= RETRY_MAX_CASE_MS), file).toBe(true); + expect(source, file).not.toMatch(/\b(?:CAPTURE_LONG_MS|PTY_MS|PTY_LONG_MS|OVERLAY_CASE_[A-Z_]+)\b/); + expect(literals.filter(ms => ms > RETRY_MAX_CASE_MS && ms < 10_000_000), file).toEqual([]); + expect(retriesForFiles([file])).toBe(1); + } + for (const file of paid.filter(file => !SHORT_CASE_RETRY_FILES.includes(file) && !FILE_RETRY_BUDGETS.some(row => row.file === file))) { + expect(retriesForFiles([file]), file).toBe(0); + } + expect(retriesForFiles([SHORT_CASE_RETRY_FILES[0]!, 'test/skill-e2e-plan.test.ts'])).toBe(0); +}); + test('registration covers exactly the seventeen demonstrated full-file retry gaps', () => { expect(Object.fromEntries(newBudgets.map(row => [row.file, row.shardMs]))).toEqual(expectedWalls); expect(new Set(FILE_RETRY_BUDGETS.map(row => row.file)).size).toBe(19); @@ -86,12 +121,17 @@ test('source allowances retain all captures, cases, and finalization grace', () ]); }); +// Case-sharded files plan one registered case per shard (`#`). +const plannedKey = (file: string) => CASE_SHARDED_FILES.includes(file) + ? `${file}#${Object.keys(E2E_TOUCHFILES).find(id => E2E_TOUCHFILES[id]!.includes(file))}` : file; + for (const row of newBudgets) { + const key = plannedKey(row.file); const manifest = () => ({ version: 1 as const, tier: 'periodic' as const, evalsAll: false, - sliceCount: 1, entries: [{ file: row.file, slice: 1, status: 'planned' as const, budget: resolvePaidShardBudget([row.file]) }] }); + sliceCount: 1, entries: [{ file: key, slice: 1, status: 'planned' as const, budget: resolvePaidShardBudget([key]) }] }); const result = (count = 1) => [{ version: 1 as const, tier: 'periodic' as const, sliceIndex: 1, sliceCount: 1, - outcomes: [{ files: [row.file], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count, - skippedTests: 0, budget: resolvePaidShardBudget([row.file]) }] }]; + outcomes: [{ files: [key], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count, + skippedTests: 0, budget: resolvePaidShardBudget([key]) }] }]; test(`${row.file}: full wall and existing retries propagate through planning`, () => { expect(retriesForFiles([row.file])).toBe(row.retries); @@ -132,8 +172,11 @@ test('fixed AUQ count remains strict while mixed-tier files keep ordinary case h [AUQ_CONSISTENCY_RETRY_BUDGET.file, 0, false], [AUQ_CONSISTENCY_RETRY_BUDGET.file, 1, true], [AUQ_CONSISTENCY_RETRY_BUDGET.file, 2, false], - ['test/skill-e2e-plan.test.ts', 5, true], - ['test/skill-e2e-plan.test.ts', 7, true], + ['test/skill-e2e-ship-docsync.test.ts', 5, true], + ['test/skill-e2e-ship-docsync.test.ts', 7, true], + // A case shard of a case-sharded registered file executes exactly its case. + [plannedKey('test/skill-e2e-plan.test.ts'), 1, true], + [plannedKey('test/skill-e2e-plan.test.ts'), 2, false], ] as const) { const budget = resolvePaidShardBudget([file]); const manifest: any = { version: 1, tier: 'periodic', evalsAll: true, sliceCount: 1, @@ -158,7 +201,7 @@ test('quality judge supervision includes the added judge without changing ordina expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS'); expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000); expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([ - ...Array(2).fill([1, 1500000, 1, 3120000]), + ...Array(2).fill([1, 1500000, 0, 1620000]), ]); for (const tier of ['gate', 'periodic'] as const) { const m = buildRunManifest({ tier, sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } }); @@ -184,7 +227,7 @@ test('detached PR fallback and release commands cover their actual default worke const prFloor = Math.ceil((Math.ceil(fullGateFiles.length / prWorkers) * 1_800_000 + fullGateFiles.reduce( (total, file) => total + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - 1_800_000), 0, )) / 1000 * 1.05); - expect(prFloor).toBe(74_981); + expect(prFloor).toBe(71_957); expect(prWall).toBe(92_820_000); expect(prWall).toBeGreaterThanOrEqual(paidShardWallUpperBoundMs(files, prWorkers) + 120_000); @@ -203,8 +246,8 @@ test('detached PR fallback and release commands cover their actual default worke )) / 1000 * 1.05)); } const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000; - expect(releaseFloors).toEqual([42_851, 37_727]); - expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(80_578); + expect(releaseFloors).toEqual([26_597, 30_797]); + expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(57_394); expect(detachedReleaseWall).toBe(116_700_000); expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000); }); @@ -217,10 +260,10 @@ const cliOptions = (step: { run: string; env?: Record }) => { return parseCliOptions(args, step.env ?? {}); }; -test('both gate executors cover the complete census without increasing aggregate workers', () => { +test('both gate executors plan the complete census and supervise every planned slice', () => { const periodic: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml')); const main: any = Bun.YAML.parse(read('.github/workflows/evals.yml')); - for (const [workflow, jobName, workers, slices] of [[main, 'eval-slices', 2, 7], [periodic, 'gate-census', 1, 7]] as const) { + for (const [workflow, jobName, prefix, skipJudges] of [[main, 'eval-slices', '', false], [periodic, 'gate-census', 'gate_', true]] as const) { const planner = workflow.jobs['plan-slices']; const executor = workflow.jobs[jobName]; const emit = planner.steps.filter((step: any) => step.run?.includes('EVALS_TIER=gate ') && step.run.includes('--emit-plan ')); @@ -230,41 +273,35 @@ test('both gate executors cover the complete census without increasing aggregate const planned = cliOptions(emit[0]), active = cliOptions(execute[0]); expect(planned.tier).toBe('gate'); expect(active.tier).toBe('gate'); - expect(active.jobs).toBe(workers); + expect(active.jobs).toBe(2); + expect(planned.jobs).toBe(active.jobs); + expect(planned.sliceBudgetMs).toBe(540_000); + expect(planned.skipJudges).toBe(skipJudges); expect(execute[0].env.EVALS_CONCURRENCY).toBe('2'); expect(executor.strategy['fail-fast']).toBe(false); - expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: slices }, (_, i) => i + 1)); - expect(planned.slices).toBe(slices); - const manifest = buildRunManifest({ tier: 'gate', sliceCount: planned.slices, evalsAll: true, env: { EVALS_ALL: '1' } }); - expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(46); - const files = manifest.entries.filter(row => row.status === 'planned').map(row => row.file); - expect(new Set(files).size).toBe(46); + expect(executor.strategy.matrix.slice).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}slices) }}`); + expect(executor['timeout-minutes']).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}timeout_minutes) }}`); + const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs, evalsAll: true, env: { EVALS_ALL: '1' }, skipJudges }); + const files = [...new Set(manifest.entries.filter(row => row.status === 'planned').map(row => shardFile(row.file)))]; expect(files).toContain('test/skill-e2e-ship-skip.test.ts'); - expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected.sort()); - const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs( - manifest.entries.filter(row => row.status === 'planned' && row.slice === slice).map(row => row.file), workers, - )); - expect(executor['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000); + expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected + .filter(file => !skipJudges || !file.startsWith('test/skill-llm-eval')).sort()); + const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder( + manifest.entries.filter(row => row.status === 'planned' && row.slice === i + 1)).map(row => row.file), active.jobs)); + expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000); + expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360); + expect(manifest.sliceCount).toBeLessThanOrEqual(executor.strategy['max-parallel']); if (jobName === 'gate-census') { - expect(Math.max(...walls)).toBe(16_440_000); - expect(executor['timeout-minutes']).toBe(352); expect(emit[0].env.EVALS_ALL).toBe('1'); - expect(executor.strategy['max-parallel']).toBe(4); - expect(executor.strategy['max-parallel'] * active.jobs).toBe(4); expect(emit[0].run).toContain('--emit-plan /tmp/gate-census-plan/manifest.json'); expect(execute[0].run).toContain('--plan /tmp/gate-census-plan/manifest.json --slice ${{ matrix.slice }}'); expect(executor.steps.some((step: any) => step.with?.name === 'gate-census-plan')).toBe(true); expect(executor.steps.filter((step: any) => step.run?.includes('--emit-plan '))).toHaveLength(0); - } else { - expect(Math.max(...walls)).toBe(12_720_000); - expect(executor['timeout-minutes']).toBe(265); - expect(executor.strategy['max-parallel']).toBe(6); - expect(executor.strategy['max-parallel'] * active.jobs).toBe(12); } } }); -test('the periodic executor supervises every actual case and retry within its CI wall', () => { +test('the periodic executor supervises every actual case and retry within its planned CI wall', () => { const workflow: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml')); const executor = workflow.jobs['eval-slices']; const emit = workflow.jobs['plan-slices'].steps.filter((step: any) => @@ -274,21 +311,19 @@ test('the periodic executor supervises every actual case and retry within its CI expect(execute).toHaveLength(1); const planned = cliOptions(emit[0]), active = cliOptions(execute[0]); expect(planned.tier).toBe('periodic'); - expect(planned.slices).toBe(7); + expect(planned.sliceBudgetMs).toBe(540_000); expect(active.jobs).toBe(2); - expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: planned.slices }, (_, i) => i + 1)); - const manifest = buildRunManifest({ tier: 'periodic', sliceCount: planned.slices, + expect(planned.jobs).toBe(active.jobs); + const manifest = buildRunManifest({ tier: 'periodic', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs, evalsAll: true, env: { EVALS_ALL: '1' } }); const census = manifest.entries.filter(row => row.status === 'planned'); - expect(census).toHaveLength(70); + expect(new Set(census.map(row => shardFile(row.file)))).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected)); expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_220_000); - const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs( - census.filter(row => row.slice === slice).map(row => row.file), active.jobs, - )); - expect(Math.max(...walls)).toBe(14_680_000); - expect(executor.strategy['max-parallel']).toBe(8); - expect(executor['timeout-minutes']).toBe(360); - expect(executor['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000); + const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder( + census.filter(row => row.slice === i + 1)).map(row => row.file), active.jobs)); + expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000); + expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360); + expect(manifest.sliceCount).toBeLessThanOrEqual(executor.strategy['max-parallel']); }); test('gate census requires all seven distinct slice results and its own reconciliation', () => { diff --git a/test/paid-run-manifest.test.ts b/test/paid-run-manifest.test.ts index 056e02210..9539df224 100644 --- a/test/paid-run-manifest.test.ts +++ b/test/paid-run-manifest.test.ts @@ -6,8 +6,9 @@ * - per-slice selector divergence → ONE planner manifest, executors consume * - hollow lanes → a slice with no artifact is a FAILURE, not an absence * - hollow shards → EVALS_ALL + exit 0 + zero executed tests ≠ pass - * - retry parity → the old matrix rows' earned `retries: 2` survive as a - * literals map, not folklore + * - retry policy → a timed-out attempt is a verdict; only short-case files retry + * - budget packing → recorded work packs into ~9-minute executors whose count + * and CI job timeout come from the plan */ import { describe, expect, test } from 'bun:test'; import * as fs from 'node:fs'; @@ -21,13 +22,18 @@ import { buildRunManifest, loadPaidTestDurations, mergePaidTestDurations, + packBySliceBudget, + estimatedSliceMs, + sliceExecutionOrder, + sliceSupervisedWallMs, + writePaidTestDurations, + CI_SETUP_ALLOWANCE_MINUTES, paidShardWallUpperBoundMs, parseCliOptions, parseRunManifest, resolvePaidShardBudget, SUPERVISED_WORKER_COUNTS, retriesForFiles, - RETRY_OVERRIDES, summarize, summaryExitCode, verifySliceResults, @@ -46,6 +52,7 @@ const outcome = (over: Partial): ShardOutcome => ({ elapsedMs: 1000, groupPid: null, executedTests: 3, + skippedTests: null, ...over, }); @@ -142,7 +149,8 @@ describe('recorded-duration slice packing', () => { test('the PR-profile file set spreads recorded time instead of stacking it', () => { const env = { EVALS_ALL: '1' }; - const discovered = Object.keys(recorded); + // Seed files only: case-shard keys (`#`) come from expansion. + const discovered = Object.keys(recorded).filter(key => !key.includes('#')); const load = (files: string[]) => files.reduce((sum, file) => sum + recorded[file], 0); const packed = lanes(buildRunManifest({ tier: 'gate', sliceCount: 6, evalsAll: true, env, discovered })).map(load); const baseline = lanes(buildRunManifest({ tier: 'gate', sliceCount: 6, evalsAll: true, env, discovered, durations: {} })).map(load); @@ -171,6 +179,88 @@ describe('recorded-duration slice packing', () => { }); }); +describe('budget slice packing', () => { + const s = (seconds: number) => seconds * 1000; + + test('best-fit packs recorded work under the budget; unknown and over-budget work gets its own runner', () => { + const recorded = { 'test/a.test.ts': s(500), 'test/b.test.ts': s(300), 'test/c.test.ts': s(240), 'test/d.test.ts': s(100), 'test/long.test.ts': s(900) }; + const files = [...Object.keys(recorded), 'test/unknown.test.ts']; + const plan = packBySliceBudget(files, s(540), 2, recorded); + expect(plan.slices.flat().sort()).toEqual([...files].sort()); + expect(plan.estimates['test/unknown.test.ts']).toBe(s(540)); + for (const [index, slice] of plan.slices.entries()) { + expect(plan.estimatedSliceMs[index]).toBe(estimatedSliceMs(slice, file => plan.estimates[file]!, 2)); + if (slice.length > 1) expect(plan.estimatedSliceMs[index]).toBeLessThanOrEqual(s(540)); + } + expect(plan.slices).toContainEqual(['test/long.test.ts']); + expect(plan.slices.find(slice => slice.includes('test/unknown.test.ts'))!.length).toBeLessThanOrEqual(2); + // Deterministic regardless of discovery order. + expect(packBySliceBudget([...files].reverse(), s(540), 2, recorded)).toEqual(plan); + const worst = Math.max(...plan.slices.map(slice => sliceSupervisedWallMs(slice, 2))); + expect(plan.ciTimeoutMinutes).toBe(Math.ceil(worst / 60_000) + CI_SETUP_ALLOWANCE_MINUTES); + expect(packBySliceBudget([], s(540), 2, recorded)).toMatchObject({ slices: [[]], ciTimeoutMinutes: CI_SETUP_ALLOWANCE_MINUTES }); + }); + + test('overlays keep one final slice at their one-at-a-time admission', () => { + const overlays = ['test/skill-e2e-overlay-harness-a.test.ts', 'test/skill-e2e-overlay-harness-b.test.ts']; + const plan = packBySliceBudget(['test/a.test.ts', ...overlays], s(540), 2, { [overlays[0]!]: s(100), [overlays[1]!]: s(200), 'test/a.test.ts': s(10) }); + expect(plan.slices.at(-1)).toEqual([overlays[1], overlays[0]]); + expect(plan.estimatedSliceMs.at(-1)).toBe(s(300)); + }); + + test('live census plans: multi-file slices stay within the budget and the executor runs longest first', () => { + for (const tier of ['gate', 'periodic'] as const) { + const manifest = buildRunManifest({ tier, sliceBudgetMs: s(540), jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } }); + expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest); + for (let slice = 1; slice <= manifest.sliceCount; slice++) { + const entries = sliceExecutionOrder(manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice)); + const estimate = manifest.plan!.estimatedSliceMs[slice - 1]!; + if (entries.length > 1 && !entries.some(entry => entry.file.includes('overlay-harness'))) expect(estimate, `${tier} slice ${slice}`).toBeLessThanOrEqual(s(540)); + expect(entries.map(entry => entry.estimatedMs)).toEqual([...entries.map(entry => entry.estimatedMs!)].sort((a, b) => b - a)); + } + } + }); + + test('plans fail closed on malformed metadata, mixed modes, and a worker count the plan did not supervise', () => { + const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: s(540), jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } }); + for (const broken of [ + { ...manifest, plan: { ...manifest.plan!, estimatedSliceMs: [] } }, + { ...manifest, plan: { ...manifest.plan!, jobs: 0 } }, + { ...manifest, entries: manifest.entries.map(entry => ({ ...entry, estimatedMs: undefined })) }, + ]) expect(() => parseRunManifest(JSON.stringify(broken))).toThrow('slice plan malformed'); + expect(() => buildRunManifest({ tier: 'gate', sliceCount: 2, sliceBudgetMs: s(540), jobs: 2, evalsAll: true })).toThrow('exactly one'); + expect(() => buildRunManifest({ tier: 'gate', sliceBudgetMs: s(540), evalsAll: true })).toThrow('explicit positive --jobs'); + expect(() => parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540'], {})).toThrow('explicit --jobs'); + expect(() => parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540', '--jobs', '2', '--slices', '3'], {})).toThrow('exactly one'); + expect(parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540'], { EVALS_JOBS: '2' }).sliceBudgetMs).toBe(s(540)); + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-plan-jobs-')); + try { + const planPath = path.join(dir, 'manifest.json'); + fs.writeFileSync(planPath, JSON.stringify(manifest)); + const result = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--plan', planPath, '--slice', '1', '--list', '--jobs', '1'], + { cwd: ROOT, encoding: 'utf8', timeout: 10_000, env: { PATH: path.dirname(process.execPath), HOME: dir, EVALS_TIER: 'gate' } }); + expect(result.status).toBe(1); + expect(result.stderr).toContain('manifest was packed for 2 worker(s) per slice'); + } finally { fs.rmSync(dir, { recursive: true, force: true }); } + }); + + test('the duration seed is per tier and a report rewrite keeps the other tiers', () => { + const gate = loadPaidTestDurations(ROOT, 'gate'); + const periodic = loadPaidTestDurations(ROOT, 'periodic'); + expect(gate['test/skill-e2e-plan.test.ts']).not.toBe(periodic['test/skill-e2e-plan.test.ts']); + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-durations-')); + try { + fs.mkdirSync(path.join(dir, 'scripts')); + writePaidTestDurations('periodic', { 'test/p.test.ts': 2_000 }, dir); + writePaidTestDurations('gate', { 'test/g.test.ts': 3_000 }, dir); + expect(loadPaidTestDurations(dir, 'gate')).toEqual({ 'test/g.test.ts': 3_000 }); + expect(loadPaidTestDurations(dir, 'periodic')).toEqual({ 'test/p.test.ts': 2_000 }); + fs.writeFileSync(path.join(dir, 'scripts/paid-test-durations.json'), JSON.stringify({ version: 1, durations: { 'test/g.test.ts': 1 } })); + expect(loadPaidTestDurations(dir, 'gate')).toEqual({}); + } finally { fs.rmSync(dir, { recursive: true, force: true }); } + }); +}); + describe('manifest executor scope', () => { test('list-only validates and prints the selected manifest slice without launching tests or writing results', () => { const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-manifest-list-')); @@ -184,8 +274,8 @@ test('local launch sentinel', () => writeFileSync(${JSON.stringify(receipt)}, 't version: 1, tier: 'gate', evalsAll: true, sliceCount: 3, selectionReason: 'local list-only fixture', entries: [ { file, slice: 1, status: 'planned' }, - { file: 'test/skill-e2e-plan.test.ts', slice: 2, status: 'planned', - budget: resolvePaidShardBudget(['test/skill-e2e-plan.test.ts']) }, + { file: 'test/skill-e2e-plan.test.ts#plan-ceo-review', slice: 2, status: 'planned', + budget: resolvePaidShardBudget(['test/skill-e2e-plan.test.ts#plan-ceo-review']) }, ], }; const manifestPath = path.join(dir, 'manifest.json'); @@ -327,7 +417,7 @@ describe('slice-result reconciliation (report)', () => { tier: 'gate', sliceIndex: index, sliceCount: 2, - outcomes: files.map((f) => ({ files: [f], status, exitCode: 0, elapsedMs: 5, executedTests: 2 })), + outcomes: files.map((f) => ({ files: [f], status, exitCode: 0, elapsedMs: 5, executedTests: 2, skippedTests: 0 })), }); test('all slices present and passing → ok', () => { @@ -397,25 +487,24 @@ describe('hollow-shard guard', () => { }); describe('retry parity', () => { - test('registered native workflows preserve main retry policy while overlay attempts stay isolated', () => { + test('registered native workflows follow the retry rule while overlay attempts stay isolated', () => { + // A 25-minute case is past RETRY_MAX_CASE_MS: its timed-out attempt is the verdict. const native = 'test/skill-e2e-plan-ceo-split-overflow.test.ts'; - expect(retriesForFiles([native])).toBe(1); - expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(1); - expect(buildPaidShardArgs([native], 1_800_000, 2, retriesForFiles([native])).join(' ')).toContain('--retry 1'); + expect(retriesForFiles([native])).toBe(0); + expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(0); + expect(buildPaidShardArgs([native], 1_800_000, 2, retriesForFiles([native])).join(' ')).toContain('--retry 0'); const overlay = 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts'; expect(retriesForFiles([overlay])).toBe(0); }); - test('overrides exist only for the files whose matrix rows earned them, and each names a real file', () => { - expect(Object.keys(RETRY_OVERRIDES).sort()).toEqual([ - 'test/skill-e2e-office-hours-auto-mode.test.ts', - 'test/skill-e2e-plan-mode-no-op.test.ts', - 'test/skill-e2e-workflow.test.ts', - ]); - for (const file of Object.keys(RETRY_OVERRIDES)) { - expect(fs.existsSync(path.join(ROOT, file)), `stale RETRY_OVERRIDES entry: ${file}`).toBe(true); + test('the matrix-era earned retries now follow the timeout-is-a-verdict rule, and each names a real file', () => { + // These three old matrix rows earned `retries: 2`; every one has a + // CAPTURE_LONG case, so a timed-out attempt is now their verdict. + for (const file of ['test/skill-e2e-office-hours-auto-mode.test.ts', 'test/skill-e2e-plan-mode-no-op.test.ts', 'test/skill-e2e-workflow.test.ts']) { + expect(fs.existsSync(path.join(ROOT, file)), `stale retry parity entry: ${file}`).toBe(true); + expect(retriesForFiles([file])).toBe(0); } - expect(retriesForFiles(['test/skill-e2e-workflow.test.ts'])).toBe(2); - expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(1); + expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(0); + expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1); expect(buildPaidShardArgs(['x'], 1000, 4, 2)).toContain('2'); expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 1'); }); diff --git a/test/paid-shards.test.ts b/test/paid-shards.test.ts index 8760d7947..e40239457 100644 --- a/test/paid-shards.test.ts +++ b/test/paid-shards.test.ts @@ -13,6 +13,7 @@ import { describe, test, expect } from 'bun:test'; import * as fs from 'fs'; import * as os from 'os'; import * as path from 'path'; +import { E2E_TIERS, E2E_TOUCHFILES } from './helpers/touchfiles'; const ROOT = path.resolve(import.meta.dir, '..'); import { @@ -31,7 +32,21 @@ import { summarize, summaryExitCode, tierSkipReason, + marathonSkipReason, + CASE_SHARDED_FILES, + CASE_TEST_NAMES, + caseTestNamePattern, + expandCaseShards, + fileCaseRegistration, + resolvePaidShardBudget, + retriesForFiles, + shardCaseId, + shardFile, + shardSlug, + verifySliceResults, + selectPaidTestFiles, buildRunManifest, + parseRunManifest, type ShardOutcome, } from '../scripts/test-paid-shards'; @@ -117,7 +132,105 @@ describe('tier lane skip (B5)', () => { } const workflow = fs.readFileSync(path.join(ROOT, '.github/workflows/evals-periodic.yml'), 'utf8'); expect(workflow.match(/--skip-judges/g)).toHaveLength(1); - expect(workflow).toMatch(/--tier gate --emit-plan \/tmp\/gate-census-plan\/manifest\.json --slices 7 --skip-judges/); + expect(workflow).toMatch(/--tier gate --emit-plan \/tmp\/gate-census-plan\/manifest\.json --slice-budget 540 --jobs 2 --skip-judges/); + }); +}); + +describe('marathon tier lane', () => { + const file = 'test/skill-e2e-sample.test.ts'; + const reg = { 'sample-gate': [file], 'sample-long': [file] } as Record; + const tiers = { 'sample-gate': 'gate', 'sample-long': 'marathon' }; + + test('a marathon-declared file never enters the gate or periodic lane', () => { + const source = "const describeE2E = describeE2ETier('marathon');"; + expect(classifyPaidTestFile(source, 'marathon')).toEqual({ included: true, reason: "declares tier 'marathon'" }); + for (const tier of ['gate', 'periodic'] as const) { + expect(classifyPaidTestFile(source, tier)).toEqual({ included: false, reason: "declares tier 'marathon' only" }); + } + expect(marathonSkipReason(file, source, {}, {})).toBeNull(); + }); + + test('marathon selects positively: only declared files or files registering a marathon case', () => { + expect(marathonSkipReason(file, "testIfSelected('sample-long', async () => {});", reg, tiers)).toBeNull(); + expect(marathonSkipReason(file, "testIfSelected(name, async () => {});", reg, tiers)).toBeNull(); + const gateOnly = { 'sample-gate': [file] }; + for (const source of ["testIfSelected('sample-gate', async () => {});", "testIfSelected(name, async () => {});", '']) + expect(marathonSkipReason(file, source, gateOnly, tiers)).toBe('skipped: declares no marathon tier and registers no marathon case'); + const periodic = "const describeE2E = describeE2ETier('periodic');"; + expect(classifyPaidTestFile(periodic, 'marathon')).toEqual({ included: false, reason: "declares tier 'periodic' only" }); + }); + + test('a registered marathon case keeps its gate sibling scheduled in the gate lane', () => { + const source = "testIfSelected('sample-gate', async () => {}); testIfSelected('sample-long', async () => {});"; + expect(tierSkipReason(file, source, 'gate', reg, tiers)).toBeNull(); + expect(tierSkipReason(file, source, 'periodic', reg, tiers)).toBe('skipped: no E2E_TIERS id has tier periodic'); + }); + + test('the live marathon lane only plans files that carry marathon work, never the LLM judges', () => { + const { selected, excluded } = selectPaidTestFiles(collectPaidTestFiles(), 'marathon'); + expect(selected).not.toContain('test/skill-llm-eval.test.ts'); + for (const file of selected) { + const source = fs.readFileSync(path.join(ROOT, file), 'utf8'); + expect(marathonSkipReason(file, source), file).toBeNull(); + } + expect(selected.length + excluded.length).toBe(collectPaidTestFiles().length); + }); +}); + +describe('case-sharded files', () => { + // Paid cases only: `if (!evalsEnabled) test(...)` blocks are free checks. + const caseNames = (source: string) => [...source.matchAll( + /(? match[2]!); + + for (const file of CASE_SHARDED_FILES) { + test(`${file}: every Bun case is a registered E2E case, so every case gets a shard`, () => { + const source = fs.readFileSync(path.join(ROOT, file), 'utf8'); + const { registered, known } = fileCaseRegistration(file, source); + expect(known).toBe(true); + expect(caseNames(source).sort()).toEqual(registered.map(id => CASE_TEST_NAMES[id] ?? id).sort()); + const keys = (['gate', 'periodic', 'marathon'] as const).flatMap(tier => expandCaseShards([file], tier)); + expect(keys.map(key => shardCaseId(key)).sort()).toEqual([...registered].sort()); + for (const key of keys) expect(shardFile(key)).toBe(file); + }); + } + + test('a case key runs exactly its case: exact name pattern, own eval slug, per-case supervision', () => { + const pattern = new RegExp(caseTestNamePattern(['design-review-detector-shim'])); + expect(pattern.test('Design review detector shim E2E design-review-detector-shim')).toBe(true); + expect(pattern.test('Design review detector shim E2E design-review-detector-shim-dom')).toBe(false); + expect(new RegExp(caseTestNamePattern(['plan-review-report'])).test('Plan Review Report E2E /plan-eng-review writes GSTACK REVIEW REPORT to plan file')).toBe(true); + const key = 'test/skill-e2e-plan.test.ts#plan-ceo-review'; + expect(shardSlug([key])).toBe('skill-e2e-plan--plan-ceo-review'); + expect(shardSlug([key])).not.toBe(shardSlug(['test/skill-e2e-plan.test.ts#plan-eng-review'])); + expect(retriesForFiles([key])).toBe(retriesForFiles(['test/skill-e2e-plan.test.ts'])); + const whole = resolvePaidShardBudget(['test/skill-e2e-plan.test.ts']); + const one = resolvePaidShardBudget([key]); + expect(one.policyId).toBe(whole.policyId); + expect(one.timeoutMs).toBeLessThan(whole.timeoutMs); + expect(planPaidShards([key, 'test/skill-e2e-plan.test.ts#plan-eng-review', 'test/a.test.ts'], { maxFilesPerShard: 3 })) + .toEqual([['test/a.test.ts'], [key], ['test/skill-e2e-plan.test.ts#plan-eng-review']]); + }); + + test('manifests plan each case once and results must execute exactly that case', () => { + const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: 540_000, jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } }); + const planned = manifest.entries.filter(entry => entry.status === 'planned'); + for (const file of CASE_SHARDED_FILES) { + expect(planned.some(entry => entry.file === file)).toBe(false); + const gateIds = Object.keys(E2E_TOUCHFILES).filter(id => E2E_TOUCHFILES[id]!.includes(file) && E2E_TIERS[id] === 'gate'); + expect(planned.filter(entry => shardFile(entry.file) === file).map(entry => shardCaseId(entry.file)).sort()).toEqual(gateIds.sort()); + } + const results = Array.from({ length: manifest.sliceCount }, (_, i) => ({ version: 1 as const, tier: 'gate' as const, sliceIndex: i + 1, sliceCount: manifest.sliceCount, + outcomes: planned.filter(entry => entry.slice === i + 1).map(entry => ({ files: [entry.file], status: 'passed' as const, exitCode: 0, elapsedMs: 1, + executedTests: shardCaseId(entry.file) ? 3 : 1, skippedTests: shardCaseId(entry.file) ? 2 : 0, ...(entry.budget ? { budget: entry.budget } : {}) })) })); + expect(verifySliceResults(manifest, results).problems.filter(problem => problem.includes('Case shard'))).toEqual([]); + const empty = structuredClone(results); + const victim = empty.flatMap(result => result.outcomes).find(outcome => shardCaseId(outcome.files[0]!))!; + victim.skippedTests = victim.executedTests; + expect(verifySliceResults(manifest, empty).problems).toContain(`Case shard must execute exactly its one case: ${victim.files[0]}`); + const whole: any = structuredClone(manifest); + whole.entries.push({ file: CASE_SHARDED_FILES[0], slice: 1, status: 'planned', estimatedMs: 1 }); + expect(() => parseRunManifest(JSON.stringify(whole))).toThrow('one registered case per shard'); }); });