feat(evals): ~12-minute blocking paid lanes and a non-blocking marathon lane

- Planner budget mode (--slice-budget S --jobs J): recorded per-tier wall
  times pack into as many ~9-minute executors as the work needs; the plan
  records per-slice estimates and the CI job timeout (supervised worst case
  + 20 min). evals.yml and evals-periodic.yml derive matrix size and
  timeout-minutes from it; max-parallel covers every slice at once.
- Case shards: plan/design/review-army/shared-libs(-paths) run one registered
  case per process (<file>#<case id>, exact name pattern, exactly one case).
- Retry rule: a timed-out attempt is a verdict. Only files whose every case
  budget is CAPTURE tier or shorter keep one retry; walls shrink to match.
- Marathon tier: positive selection, excluded from gate/periodic planners,
  run by the new evals-marathon.yml (weekly + dispatch, fresh, own report).
- PR-lane E2E reuse of verified first-attempt passes on identical inputs;
  the report rejects reuse outside the fast PR profile.
- Duration seed from census run 36385945043, per tier and per case shard.
This commit is contained in:
garrytan committed 2026-09-29 16:19:52 +00:00
1 parent acb7bc02b1
commit 0023d011a3
18 files changed
+1773 -450

No files matched your search

+298
View File
@@ -0,0 +1,298 @@
name: Marathon Evals
# The NON-BLOCKING marathon lane: complete start-to-finish flows (tier
# 'marathon' in test/helpers/touchfiles-data.ts / describeE2ETier('marathon'))
# that take longer than a blocking lane's ~12-minute wall. They never run in
# the PR gate (evals.yml) or the weekly periodic + gate census
# (evals-periodic.yml); nothing requires this workflow, so a red marathon
# reports through its own tracking issue without gating any merge. Same engine
# and FAIL-CLOSED report as the other lanes: one planner manifest, one file per
# runner, a missing slice artifact is a failure. Always fresh: no result reuse.
on:
schedule:
- cron: '0 12 * * 6' # Saturday 12:00 UTC, clear of the Monday periodic census
workflow_dispatch:
concurrency:
group: evals-marathon
cancel-in-progress: true
env:
IMAGE: ghcr.io/${{ github.repository }}/ci
EVALS_PROFILE: full
EVALS_FRESH: "1"
EVALS_CACHE_PURPOSE: marathon
jobs:
IMAGE: ghcr.io/${{ github.repository }}/ci
EVALS_PROFILE: full
EVALS_FRESH: "1"
EVALS_CACHE_PURPOSE: periodic
jobs:
build-image:
runs-on: ubicloud-standard-8
timeout-minutes: 15
permissions:
contents: read
packages: write
outputs:
image-tag: ${{ steps.meta.outputs.tag }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- id: meta
# Keep in sync with evals.yml and evals-periodic.yml — key on Dockerfile + lockfile only
# (package.json's version field would bust the key on every ship).
# Byte-identity pinned by test/ci-image-tag-binding.test.ts.
run: echo "tag=${{ env.IMAGE }}:${{ hashFiles('.github/docker/Dockerfile.ci', 'bun.lock', 'patches/**') }}" >> "$GITHUB_OUTPUT"
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Check if image exists
id: check
run: |
if docker manifest inspect ${{ steps.meta.outputs.tag }} > /dev/null 2>&1; then
echo "exists=true" >> "$GITHUB_OUTPUT"
else
echo "exists=false" >> "$GITHUB_OUTPUT"
fi
- if: steps.check.outputs.exists == 'false'
run: cp package.json bun.lock .github/docker/ && cp -R patches .github/docker/patches
# Registry cache export needs a docker-container builder — the default
# `docker` driver hard-errors on cache-to.
- if: steps.check.outputs.exists == 'false'
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4
- if: steps.check.outputs.exists == 'false'
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7
with:
context: .github/docker
file: .github/docker/Dockerfile.ci
push: true
# Cron-triggered in the base repo only, so cache export is always safe here.
cache-from: type=registry,ref=${{ env.IMAGE }}:buildcache
cache-to: type=registry,ref=${{ env.IMAGE }}:buildcache,mode=max
tags: |
${{ steps.meta.outputs.tag }}
${{ env.IMAGE }}:latest
plan-slices:
runs-on: ubicloud-standard-8
timeout-minutes: 10
permissions:
contents: read
outputs:
slices: ${{ steps.matrix.outputs.slices }}
timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
persist-credentials: false
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
with:
bun-version: 1.4.0
# One marathon file per runner: a 1-second budget never packs two
# recorded files together.
- name: Emit run manifest (ALL marathon tests)
env:
EVALS_ALL: "1"
run: EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --emit-plan /tmp/marathon-plan/manifest.json --slice-budget 1 --jobs 1
- name: Derive the executor matrix from the plan
id: matrix
run: |
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT"
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: marathon-plan
path: /tmp/marathon-plan/manifest.json
retention-days: 30
eval-slices:
runs-on: ubicloud-standard-8
needs: [build-image, plan-slices]
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-marathon-${{ matrix.slice }}
# One marathon file per runner; the job timeout is the plan's supervised
# worst case plus 20 minutes setup/upload.
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }}
permissions:
contents: read
packages: read
container:
image: ${{ needs.build-image.outputs.image-tag }}
credentials:
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
options: --user runner
strategy:
fail-fast: false
max-parallel: 8
matrix:
slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
# Full history: files with SELF-derived selection (the LLM-judge
# map, routing) walk git at module load, and selection is
# fail-closed on git errors — a shallow checkout crashed those
# shards on the lane's first live run ("ambiguous argument
# 'main...HEAD'"). The manifest still governs WHICH shards run.
fetch-depth: 0
persist-credentials: false
- name: Fix bun temp
uses: ./.github/actions/fix-bun-temp
- name: Restore deps
uses: ./.github/actions/restore-deps
- run: bun run build
# Any slice can host a PTY test — seed + registration run
# unconditionally (idempotent; mirrors evals.yml's sliced lane). The
# register composite carries the fail-fast dangling-symlink/frontmatter
# verification loop — this lane previously LACKED it, so a moved skill
# target surfaced as a silent "Unknown command" + wedged PTY session.
- name: Seed claude interactive config
uses: ./.github/actions/seed-claude-config
with:
anthropic-api-key: ${{ secrets.ANTHROPIC_API_KEY }}
- name: Register gstack skills for PTY tests
uses: ./.github/actions/register-gstack-skills
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
name: marathon-plan
path: /tmp/marathon-plan
- name: Run marathon slice ${{ matrix.slice }}
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers
EVALS_JOBS: "1"
EVALS_CONCURRENCY: "2"
GSTACK_EVAL_DIR: /tmp/marathon-slice-results
run: EVALS_TIER=marathon bun run scripts/test-paid-shards.ts --tier marathon --plan /tmp/marathon-plan/manifest.json --slice ${{ matrix.slice }}
- name: Upload slice results
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: marathon-slice-${{ matrix.slice }}
path: /tmp/marathon-slice-results
retention-days: 90
- name: Upload native capture evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: native-captures-${{ env.EVALS_RUN_ID }}
include-hidden-files: true
path: |
~/.gstack/projects/*/e2e-runs
~/.gstack/projects/*/evals/qa-callers
~/.gstack-dev/e2e-runs
~/.gstack-dev/evals/qa-callers
if-no-files-found: ignore
retention-days: 90
- name: Upload shard logs on failure
if: failure()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: marathon-slice-${{ matrix.slice }}-logs
include-hidden-files: true
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
# runner's spool lands THERE, not /tmp — the original /tmp glob
# uploaded nothing and a red slice's diagnostics were unreachable.
path: |
/home/runner/.cache/gstack-paid-shard-*.log
/tmp/gstack-paid-shard-*.log
if-no-files-found: ignore
retention-days: 30
report:
runs-on: ubicloud-standard-2
needs: [plan-slices, eval-slices]
# !cancelled(): the report must run (and FAIL) when an executor died — a
# missing slice artifact reading as green is the class this lane kills —
# but a cancelled run stops here.
if: ${{ !cancelled() && needs.plan-slices.result == 'success' }}
timeout-minutes: 10
permissions:
contents: read
issues: write
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
persist-credentials: false
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
with:
bun-version: 1.4.0
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
name: marathon-plan
path: /tmp/marathon-report
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
pattern: marathon-slice-[0-9]*
path: /tmp/marathon-report
merge-multiple: true
- name: Reconcile slices against the manifest (fail-closed)
id: reconcile
if: always()
run: |
set +e
EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --report /tmp/marathon-report | tee /tmp/report.txt
# PIPESTATUS[0], NOT $?: the default step shell has no pipefail.
echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
# One tracking issue for the whole lane (never one per week).
- name: Upsert tracking issue on failure
if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success')
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set -euo pipefail
TITLE="Weekly marathon evals: red lane needs triage"
BODY_FILE=/tmp/issue-body.md
{
echo "Automated weekly marathon report (non-blocking lane) — run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
echo
echo "- reconciliation exit: ${{ steps.reconcile.outputs.exit }}"
echo "- marathon slices job: ${{ needs.eval-slices.result }}"
echo
echo '```'
tail -c 6000 /tmp/report.txt 2>/dev/null || echo "(no reconciliation output)"
echo '```'
} > "$BODY_FILE"
EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty')
if [ -n "$EXISTING" ]; then
gh issue comment "$EXISTING" --repo "$GITHUB_REPOSITORY" --body-file "$BODY_FILE"
echo "commented on #$EXISTING"
else
gh issue create --repo "$GITHUB_REPOSITORY" --title "$TITLE" --body-file "$BODY_FILE"
fi
- name: Fail the workflow when reconciliation failed
if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success')
run: exit 1
+42 -17
View File
@@ -4,8 +4,13 @@ name: Periodic Evals
# tests can't rot invisibly — the class where the autoplan-dual-voice E2E was
# silently broken for months until a lucky local diff selected it. Engine:
# scripts/test-paid-shards.ts (the same runner local eval:bg:periodic uses):
# one planner manifest, 6 ordinary slices plus an overlay slice, and a FAIL-CLOSED report — a slice
# whose artifact never landed is a failure, not an absence. The gate-census
# one planner manifest packed by recorded durations into as many ~9-minute
# executors as the work needs (one file, or a tightly packed group, per
# runner; overlays share one final slice), and a FAIL-CLOSED report — a slice
# whose artifact never landed is a failure, not an absence. The matrix size
# and job timeout come from the plan, so they cannot drift from the census.
# Full end-to-end flows run in the non-blocking marathon lane
# (evals-marathon.yml), never here. The gate-census
# job is the weekly EVALS_ALL backstop for the gate tier (PR lanes are
# diff-billed, so without it the full gate census might never execute
# anywhere); the hollow-shard guard (exit 0 + zero executed tests under
@@ -84,6 +89,11 @@ jobs:
timeout-minutes: 10
permissions:
contents: read
outputs:
periodic_slices: ${{ steps.periodic-matrix.outputs.slices }}
periodic_timeout_minutes: ${{ steps.periodic-matrix.outputs.timeout_minutes }}
gate_slices: ${{ steps.gate-matrix.outputs.slices }}
gate_timeout_minutes: ${{ steps.gate-matrix.outputs.timeout_minutes }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -96,7 +106,13 @@ jobs:
- name: Emit run manifest (ALL periodic tests minus reasoned excludes)
env:
EVALS_ALL: "1"
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 7
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2
- name: Derive the periodic executor matrix from the plan
id: periodic-matrix
run: |
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
@@ -107,7 +123,13 @@ jobs:
- name: Emit gate census manifest (ALL gate tests)
env:
EVALS_ALL: "1"
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 7 --skip-judges
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges
- name: Derive the gate census executor matrix from the plan
id: gate-matrix
run: |
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/gate-census-plan/manifest.json)" >> "$GITHUB_OUTPUT"
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/gate-census-plan/manifest.json)" >> "$GITHUB_OUTPUT"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
@@ -120,9 +142,10 @@ jobs:
needs: [build-image, plan-slices]
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }}
# Seven slices retain every registered case and retry. The complete
# census needs at most 244m40s per slice, plus 20 minutes setup/upload.
timeout-minutes: 360
# The planner packs ~9 minutes of recorded work per slice; the job timeout
# is its supervised worst case (every shard at its wall) plus 20 minutes
# setup/upload, computed from the same manifest the slices execute.
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}
permissions:
contents: read
packages: read
@@ -134,9 +157,11 @@ jobs:
options: --user runner
strategy:
fail-fast: false
max-parallel: 8
# Every planned slice starts at once; test/evals-workflow-wiring.test.ts
# fails when the live plan outgrows this cap.
max-parallel: 24
matrix:
slice: [1, 2, 3, 4, 5, 6, 7]
slice: ${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -174,7 +199,7 @@ jobs:
name: paid-plan
path: /tmp/paid-plan
- name: Run slice ${{ matrix.slice }}/7
- name: Run periodic slice ${{ matrix.slice }}
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
@@ -231,8 +256,8 @@ jobs:
needs: [build-image, plan-slices]
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-gate-census-${{ matrix.slice }}
# Seven slices need at most 272m each, plus 20 minutes setup/upload.
timeout-minutes: 352
# Supervised worst case of the packed plan plus 20 minutes setup/upload.
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.gate_timeout_minutes) }}
permissions:
contents: read
packages: read
@@ -243,11 +268,11 @@ jobs:
password: ${{ secrets.GITHUB_TOKEN }}
options: --user runner
strategy:
# Four file workers total, each retaining two in-file case workers.
# Two file workers per slice, each retaining two in-file case workers.
fail-fast: false
max-parallel: 4
max-parallel: 16
matrix:
slice: [1, 2, 3, 4, 5, 6, 7]
slice: ${{ fromJSON(needs.plan-slices.outputs.gate_slices) }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -272,13 +297,13 @@ jobs:
name: gate-census-plan
path: /tmp/gate-census-plan
- name: Run gate census slice ${{ matrix.slice }}/7
- name: Run gate census slice ${{ matrix.slice }}
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers
EVALS_JOBS: "1"
EVALS_JOBS: "2"
EVALS_CONCURRENCY: "2"
GSTACK_EVAL_DIR: /tmp/gate-census-results
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/gate-census-plan/manifest.json --slice ${{ matrix.slice }}
+29 -22
View File
@@ -125,6 +125,9 @@ jobs:
timeout-minutes: 10
permissions:
contents: read
outputs:
slices: ${{ steps.matrix.outputs.slices }}
timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -141,7 +144,7 @@ jobs:
if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all'
env:
EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }}
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slices 7
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2
- name: Emit validation-phase manifest
if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all'
@@ -152,23 +155,30 @@ jobs:
run: |
bun --no-install -e '
import { mkdirSync, writeFileSync } from "node:fs";
import { buildRunManifest, collectPaidTestFiles } from "./scripts/test-paid-shards.ts";
import { buildRunManifest, collectPaidTestFiles, restrictManifestSelection } from "./scripts/test-paid-shards.ts";
const phase = process.env.VALIDATION_PHASE;
if (!["quality", "cookie-quality", "behavior", "cookie-behavior"].includes(phase)) throw new Error("Invalid validation phase");
const cookieBehavior = phase === "cookie-behavior";
const discovered = phase === "cookie-quality" ? ["test/skill-llm-eval.test.ts"]
: cookieBehavior ? ["test/skill-e2e-bws.test.ts", "test/skill-e2e-qa-workflow.test.ts", "test/skill-e2e-design.test.ts", "test/skill-e2e-diagram.test.ts", "test/skill-e2e-deploy.test.ts"]
: collectPaidTestFiles().filter(file => file.startsWith("test/skill-llm-eval") === (phase === "quality"));
const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceCount: 6, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered,
const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceBudgetMs: 540000, jobs: 2, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered,
...(cookieBehavior ? { changedFiles: ["browse/src/cookie-picker-routes.ts", "browse/src/cookie-import-browser.ts", "browse/src/bun-polyfill.cjs"], env: { ...process.env, EVALS_ALL: "" } } : {}) });
if (phase === "cookie-quality") manifest.selection = { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] };
if (cookieBehavior) manifest.selection = { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] };
manifest.selectionReason = phase + " validation subset; " + manifest.selectionReason;
const subset = phase === "cookie-quality" ? { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] }
: cookieBehavior ? { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] } : null;
const restricted = subset ? restrictManifestSelection(manifest, subset, "outside the " + phase + " validation subset") : manifest;
restricted.selectionReason = phase + " validation subset; " + manifest.selectionReason;
mkdirSync("/tmp/paid-plan", { recursive: true });
writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(manifest, null, 2) + "\n");
console.log(phase + ": " + manifest.entries.filter(entry => entry.status === "planned").length + " planned shards");
writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(restricted, null, 2) + "\n");
console.log(phase + ": " + restricted.entries.filter(entry => entry.status === "planned").length + " planned shards");
'
- name: Derive the executor matrix from the plan
id: matrix
run: |
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: paid-plan
@@ -184,14 +194,11 @@ jobs:
# (image already published), but a newer push's cancel-in-progress stops
# it instead of letting a superseded run finish its paid slices first.
if: ${{ !cancelled() && needs.build-image.result == 'success' && needs.plan-slices.result == 'success' }}
# Aggregate spawn-concurrency budget: 6 slices x EVALS_JOBS=2 x
# EVALS_CONCURRENCY=2 = 24 concurrent tests lane-wide (the old matrix's
# 40-way per row queued claude session STARTUP behind 39 siblings and ate
# per-test budgets — the documented timeout-flake family). Tune with
# parity data before raising.
# The complete gate census needs at most 242 minutes per slice; keep
# 20 minutes for setup/upload without preempting configured retries.
timeout-minutes: 265
# The planner packs ~9 minutes of recorded work per slice (EVALS_JOBS=2 x
# EVALS_CONCURRENCY=2 per runner, never the old 40-way per-row fan-out
# that queued claude session STARTUP behind 39 siblings). The job timeout
# is the plan's supervised worst case plus 20 minutes setup/upload.
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }}
permissions:
contents: read
packages: read
@@ -203,9 +210,9 @@ jobs:
options: --user runner
strategy:
fail-fast: false
max-parallel: 6
max-parallel: 16
matrix:
slice: [1, 2, 3, 4, 5, 6, 7]
slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -246,7 +253,7 @@ jobs:
# Only this PR's receipts are eligible. No base-branch or cross-PR restore
# prefix; every receipt also verifies exact inputs and its original age.
- name: Restore this PR's verified judge results
- name: Restore this PR's verified judge and E2E results
if: github.event_name == 'pull_request'
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
with:
@@ -254,7 +261,7 @@ jobs:
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}
restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-
- name: Run slice ${{ matrix.slice }}/7
- name: Run slice ${{ matrix.slice }}
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
@@ -285,7 +292,7 @@ jobs:
# An unrelated failing case does not discard already verified passes.
# Failed/retried/partial attempts never become receipts in the first place.
- name: Save verified judge results for this PR
- name: Save verified judge and E2E results for this PR
if: ${{ !cancelled() && steps.receipts.outputs.present == 'true' }}
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
with:
@@ -531,7 +538,7 @@ jobs:
</details>
---
*Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → 6 executors → fail-closed report). Reused scores retain their original provenance and expiry.*"
*Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → duration-packed executors → fail-closed report). Reused results retain their original provenance and expiry.*"
if [ "$FAILED" -gt 0 ]; then
FAILURES=""
+175 -41
View File
@@ -1,44 +1,178 @@
{
"version": 1,
"recordedAt": "2026-09-28T23:00:00Z",
"durations": {
"test/skill-e2e-ask-user-question-format-compliance.test.ts": 67000,
"test/skill-e2e-bws.test.ts": 99000,
"test/skill-e2e-coverage-audit.test.ts": 60000,
"test/skill-e2e-cso.test.ts": 220000,
"test/skill-e2e-deploy.test.ts": 496000,
"test/skill-e2e-design.test.ts": 228000,
"test/skill-e2e-diagram.test.ts": 25000,
"test/skill-e2e-docsync-spawned.test.ts": 49000,
"test/skill-e2e-hermetic-canary.test.ts": 5000,
"test/skill-e2e-investigate-owned-completion.test.ts": 50000,
"test/skill-e2e-investigate-owned-termination.test.ts": 49000,
"test/skill-e2e-learnings.test.ts": 32000,
"test/skill-e2e-office-hours-auto-mode.test.ts": 61000,
"test/skill-e2e-plan-ceo-finding-floor.test.ts": 233000,
"test/skill-e2e-plan-ceo-plan-mode.test.ts": 35000,
"test/skill-e2e-plan-design-with-ui.test.ts": 435000,
"test/skill-e2e-plan-devex-finding-floor.test.ts": 187000,
"test/skill-e2e-plan-devex-plan-mode.test.ts": 103000,
"test/skill-e2e-plan-mode-no-op.test.ts": 206000,
"test/skill-e2e-plan-tune.test.ts": 58000,
"test/skill-e2e-plan.test.ts": 312000,
"test/skill-e2e-qa-workflow.test.ts": 437000,
"test/skill-e2e-retro.test.ts": 139000,
"test/skill-e2e-review-army.test.ts": 568000,
"test/skill-e2e-review-attribution.test.ts": 80000,
"test/skill-e2e-review.test.ts": 135000,
"test/skill-e2e-session-intelligence.test.ts": 47000,
"test/skill-e2e-shared-libs-paths.test.ts": 713000,
"test/skill-e2e-shared-libs.test.ts": 662000,
"test/skill-e2e-ship-docsync.test.ts": 148000,
"test/skill-e2e-ship-hook-consent.test.ts": 72000,
"test/skill-e2e-ship-hook-refresh.test.ts": 66000,
"test/skill-e2e-skillify.test.ts": 156000,
"test/skill-e2e-third-party-actions.test.ts": 75000,
"test/skill-e2e-triage.test.ts": 49000,
"test/skill-e2e-workflow.test.ts": 234000,
"test/skill-llm-eval.test.ts": 103000,
"test/skill-routing-e2e.test.ts": 1000
"version": 2,
"recordedAt": "2026-09-28T09:05:00Z",
"source": "periodic census run 36385945043 (eval-slices = periodic, gate-census = gate); timed-out shards record their wall, self-skipping shards 1s; case shards (<file>#<case>) from the Bun per-case walls in that run (gate census log) and the eval JSON slowest cases (periodic)",
"tiers": {
"gate": {
"test/llm-judge-recommendation.test.ts": 1000,
"test/skill-e2e-ask-user-question-format-compliance.test.ts": 60000,
"test/skill-e2e-autoplan-dual-voice.test.ts": 1000,
"test/skill-e2e-bws.test.ts": 82000,
"test/skill-e2e-context-skills.test.ts": 1000,
"test/skill-e2e-coverage-audit.test.ts": 46000,
"test/skill-e2e-cso.test.ts": 251000,
"test/skill-e2e-deploy.test.ts": 427000,
"test/skill-e2e-design.test.ts": 261000,
"test/skill-e2e-design.test.ts#design-review-detector-shim": 51000,
"test/skill-e2e-design.test.ts#design-review-detector-shim-dom": 108000,
"test/skill-e2e-design.test.ts#design-review-plugin-handoff": 110000,
"test/skill-e2e-design.test.ts#plan-design-review-no-ui-scope": 43000,
"test/skill-e2e-diagram.test.ts": 32000,
"test/skill-e2e-docsync-spawned.test.ts": 51000,
"test/skill-e2e-first-task-scaffold.test.ts": 1000,
"test/skill-e2e-gbrain-roundtrip-local.test.ts": 1000,
"test/skill-e2e-hermetic-canary.test.ts": 8000,
"test/skill-e2e-investigate-owned-completion.test.ts": 45000,
"test/skill-e2e-investigate-owned-termination.test.ts": 44000,
"test/skill-e2e-ios-device.test.ts": 1000,
"test/skill-e2e-learnings.test.ts": 35000,
"test/skill-e2e-office-hours-auto-mode.test.ts": 67000,
"test/skill-e2e-office-hours-brain-writeback.test.ts": 1000,
"test/skill-e2e-office-hours-phase4.test.ts": 1000,
"test/skill-e2e-office-hours.test.ts": 1000,
"test/skill-e2e-plan-ceo-finding-floor.test.ts": 516000,
"test/skill-e2e-plan-ceo-plan-mode.test.ts": 39000,
"test/skill-e2e-plan-decision-classification.test.ts": 1000,
"test/skill-e2e-plan-design-with-ui.test.ts": 500000,
"test/skill-e2e-plan-devex-finding-floor.test.ts": 233000,
"test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 1000,
"test/skill-e2e-plan-devex-plan-mode.test.ts": 85000,
"test/skill-e2e-plan-format.test.ts": 1000,
"test/skill-e2e-plan-mode-no-op.test.ts": 206000,
"test/skill-e2e-plan-prosons.test.ts": 1000,
"test/skill-e2e-plan-tune.test.ts": 60000,
"test/skill-e2e-plan.test.ts": 251000,
"test/skill-e2e-plan.test.ts#codex-offered-ceo-review": 55000,
"test/skill-e2e-plan.test.ts#codex-offered-design-review": 59000,
"test/skill-e2e-plan.test.ts#codex-offered-eng-review": 58000,
"test/skill-e2e-plan.test.ts#codex-offered-office-hours": 54000,
"test/skill-e2e-plan.test.ts#office-hours-spec-review": 34000,
"test/skill-e2e-plan.test.ts#plan-ceo-review-benefits": 52000,
"test/skill-e2e-plan.test.ts#plan-review-report": 51000,
"test/skill-e2e-qa-bugs.test.ts": 1000,
"test/skill-e2e-qa-workflow.test.ts": 398000,
"test/skill-e2e-retro.test.ts": 152000,
"test/skill-e2e-review-army.test.ts": 520000,
"test/skill-e2e-review-army.test.ts#review-army-delivery-audit": 48000,
"test/skill-e2e-review-army.test.ts#review-army-json-findings": 24000,
"test/skill-e2e-review-army.test.ts#review-army-migration-safety": 140000,
"test/skill-e2e-review-army.test.ts#review-army-perf-n-plus-one": 244000,
"test/skill-e2e-review-army.test.ts#review-army-quality-score": 64000,
"test/skill-e2e-review-attribution.test.ts": 87000,
"test/skill-e2e-review.test.ts": 129000,
"test/skill-e2e-session-intelligence.test.ts": 59000,
"test/skill-e2e-shared-libs-paths.test.ts": 680000,
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-index-flags": 201000,
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-path-eligibility": 252000,
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-prior-coverage": 227000,
"test/skill-e2e-shared-libs.test.ts": 658000,
"test/skill-e2e-shared-libs.test.ts#shared-libs-read-only": 230000,
"test/skill-e2e-shared-libs.test.ts#shared-libs-review-lifecycle": 278000,
"test/skill-e2e-shared-libs.test.ts#shared-libs-review-revalidation": 427000,
"test/skill-e2e-shared-libs.test.ts#shared-libs-unsupported-git": 168000,
"test/skill-e2e-ship-docsync.test.ts": 129000,
"test/skill-e2e-ship-hook-consent.test.ts": 63000,
"test/skill-e2e-ship-hook-refresh.test.ts": 70000,
"test/skill-e2e-skillify.test.ts": 188000,
"test/skill-e2e-third-party-actions.test.ts": 70000,
"test/skill-e2e-triage.test.ts": 80000,
"test/skill-e2e-workflow.test.ts": 300000,
"test/skill-llm-eval.test.ts": 354000,
"test/skill-routing-e2e.test.ts": 1000
},
"periodic": {
"test/carve-section-loading-browse.test.ts": 109000,
"test/carve-section-loading-codex.test.ts": 118000,
"test/carve-section-loading-design-consultation.test.ts": 349000,
"test/carve-section-loading-design-html.test.ts": 259000,
"test/carve-section-loading-design-shotgun.test.ts": 194000,
"test/carve-section-loading-document-release.test.ts": 95000,
"test/carve-section-loading-land-and-deploy.test.ts": 185000,
"test/carve-section-loading-plan-design-review.test.ts": 284000,
"test/carve-section-loading-plan-devex-review.test.ts": 396000,
"test/carve-section-loading-plan-eng-review.test.ts": 301000,
"test/carve-section-loading-qa.test.ts": 114000,
"test/carve-section-loading-retro.test.ts": 230000,
"test/carve-section-loading-review.test.ts": 174000,
"test/carve-section-loading-setup-gbrain.test.ts": 91000,
"test/carve-section-loading-spec.test.ts": 150000,
"test/codex-e2e-recommendation-substance.test.ts": 1000,
"test/codex-e2e-shared-libs.test.ts": 1000,
"test/codex-e2e-sol-scope.test.ts": 1000,
"test/codex-e2e.test.ts": 1000,
"test/llm-judge-recommendation.test.ts": 15000,
"test/skill-e2e-arm-benchmark.test.ts": 76000,
"test/skill-e2e-aside.test.ts": 1000,
"test/skill-e2e-auq-consistency.test.ts": 75000,
"test/skill-e2e-auq-matrix.test.ts": 320000,
"test/skill-e2e-auq-verbose-vs-carved-ab.test.ts": 66000,
"test/skill-e2e-auto-decide-preserved.test.ts": 113000,
"test/skill-e2e-autoplan-dual-voice.test.ts": 532000,
"test/skill-e2e-benchmark-providers.test.ts": 9000,
"test/skill-e2e-bws.test.ts": 1000,
"test/skill-e2e-context-skills.test.ts": 181000,
"test/skill-e2e-coverage-audit.test.ts": 1000,
"test/skill-e2e-cso.test.ts": 358000,
"test/skill-e2e-deploy.test.ts": 1000,
"test/skill-e2e-design.test.ts": 817000,
"test/skill-e2e-design.test.ts#design-consultation-core": 204000,
"test/skill-e2e-design.test.ts#design-html-slop-gate": 205000,
"test/skill-e2e-design.test.ts#plan-design-review-plan-mode": 283000,
"test/skill-e2e-diagram.test.ts": 166000,
"test/skill-e2e-first-task-scaffold.test.ts": 9000,
"test/skill-e2e-gbrain-roundtrip-local.test.ts": 1000,
"test/skill-e2e-health.test.ts": 165000,
"test/skill-e2e-hermetic-canary.test.ts": 1000,
"test/skill-e2e-ios-device.test.ts": 1000,
"test/skill-e2e-learnings.test.ts": 1000,
"test/skill-e2e-office-hours-brain-writeback.test.ts": 174000,
"test/skill-e2e-office-hours-phase4.test.ts": 60000,
"test/skill-e2e-office-hours-section-loading.test.ts": 1200000,
"test/skill-e2e-office-hours.test.ts": 119000,
"test/skill-e2e-outside-plan-disabled.test.ts": 43000,
"test/skill-e2e-outside-voice.test.ts": 1000,
"test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts": 81000,
"test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts": 66000,
"test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts": 42000,
"test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts": 278000,
"test/skill-e2e-plan-ceo-mode-routing.test.ts": 575000,
"test/skill-e2e-plan-ceo-review-section-loading.test.ts": 604000,
"test/skill-e2e-plan-ceo-split-overflow.test.ts": 1332000,
"test/skill-e2e-plan-decision-classification.test.ts": 122000,
"test/skill-e2e-plan-design-finding-floor.test.ts": 175000,
"test/skill-e2e-plan-design-plan-mode.test.ts": 117000,
"test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 83000,
"test/skill-e2e-plan-eng-finding-floor.test.ts": 282000,
"test/skill-e2e-plan-eng-multi-finding-batching.test.ts": 734000,
"test/skill-e2e-plan-eng-plan-mode.test.ts": 91000,
"test/skill-e2e-plan-format.test.ts": 282000,
"test/skill-e2e-plan-prosons.test.ts": 180000,
"test/skill-e2e-plan-tune.test.ts": 1000,
"test/skill-e2e-plan.test.ts": 824000,
"test/skill-e2e-plan.test.ts#plan-ceo-review": 123000,
"test/skill-e2e-plan.test.ts#plan-ceo-review-selective": 244000,
"test/skill-e2e-plan.test.ts#plan-eng-review-artifact": 153000,
"test/skill-e2e-qa-bugs.test.ts": 283000,
"test/skill-e2e-qa-workflow.test.ts": 239000,
"test/skill-e2e-retro.test.ts": 114000,
"test/skill-e2e-review-army.test.ts": 511000,
"test/skill-e2e-review-army.test.ts#review-army-consensus": 278000,
"test/skill-e2e-review-army.test.ts#review-army-simplification": 144000,
"test/skill-e2e-review-attribution.test.ts": 1000,
"test/skill-e2e-review.test.ts": 131000,
"test/skill-e2e-session-intelligence.test.ts": 1000,
"test/skill-e2e-setup-gbrain-bad-token.test.ts": 49000,
"test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts": 119000,
"test/skill-e2e-setup-gbrain-remote.test.ts": 110000,
"test/skill-e2e-shared-libs-periodic.test.ts": 415000,
"test/skill-e2e-ship-section-loading.test.ts": 379000,
"test/skill-e2e-skillify.test.ts": 56000,
"test/skill-e2e-sync-gbrain-readiness.test.ts": 74000,
"test/skill-e2e-third-party-actions.test.ts": 1000,
"test/skill-e2e-triage.test.ts": 1000,
"test/skill-e2e-workflow.test.ts": 1000,
"test/skill-llm-eval.test.ts": 402000,
"test/skill-routing-e2e.test.ts": 57000
}
}
}
+464 -115
View File
@@ -67,12 +67,15 @@ import {
} from './test-strict-output';
import { PAID_TEST_GLOBS, isPaidTestFile } from '../test/helpers/paid-test-set';
import { PERIODIC_CI_EXCLUDE } from '../test/helpers/periodic-exclude-data';
import { FILE_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets';
import { FILE_RETRY_BUDGETS, SHORT_CASE_RETRY_FILES, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets';
import { getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile, evalEntryOutcome } from '../test/helpers/eval-store';
import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-review';
import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight';
import { OVERLAY_MIN_FILE_WALL_MS } from '../test/helpers/overlay-case-policy';
import { PR_PROFILE_CASE_IDS, PR_PROFILE_FILES, packageChangeOnlyVersion, selectPrProfile, type PrProfileSelection } from './test-pr-profile';
import { e2eReuseLaneProblem, prepareE2EShardReuse } from './e2e-shard-reuse';
type E2EShardReuse = NonNullable<ReturnType<typeof prepareE2EShardReuse>>;
import {
detectBaseBranch,
getChangedFiles,
@@ -88,7 +91,8 @@ export { PERIODIC_CI_EXCLUDE };
const ROOT = path.resolve(import.meta.dir, '..');
export type PaidTier = 'gate' | 'periodic';
export type PaidTier = 'gate' | 'periodic' | 'marathon';
export const PAID_TIERS: readonly PaidTier[] = ['gate', 'periodic', 'marathon'];
export type PaidProfile = 'pr' | 'full';
export interface PaidCaseSelection {
@@ -117,6 +121,64 @@ export function isOverlayTestFile(file: string): boolean {
return /^skill-e2e-overlay-harness-.+\.test\.ts$/.test(path.basename(normalizeRelativePath(file)));
}
/**
* Files whose cases run in separate processes, one shard per registered E2E
* case (`<file>#<case id>`): the file's lane wall exceeds one runner's budget
* while every case is short. Separate processes also give each case its own
* SDK semaphore, so shared-libs(-paths) capture waves never queue inside a
* sibling case's wall (the reason paths runs test.serial in one process).
* Every case must be a registered, literal E2E id whose Bun test name is the
* id or its CASE_TEST_NAMES label (test/paid-shards.test.ts scans the sources).
*/
export const CASE_SHARDED_FILES: readonly string[] = [
'test/skill-e2e-design.test.ts',
'test/skill-e2e-plan.test.ts',
'test/skill-e2e-review-army.test.ts',
'test/skill-e2e-shared-libs-paths.test.ts',
'test/skill-e2e-shared-libs.test.ts',
];
/** Bun test names that differ from their E2E id. */
export const CASE_TEST_NAMES: Record<string, string> = {
'plan-review-report': '/plan-eng-review writes GSTACK REVIEW REPORT to plan file',
'auq-format-gate': "/plan-ceo-review's first AskUserQuestion is a compliant decision brief (7/7 + substance)",
};
const CASE_KEY_SEPARATOR = '#';
/** The test file behind a shard key (`<file>` or `<file>#<case id>`). */
export function shardFile(key: string): string {
return normalizeRelativePath(key).split(CASE_KEY_SEPARATOR)[0]!;
}
/** The E2E case id of a case shard key, else null. */
export function shardCaseId(key: string): string | null {
const [, id] = normalizeRelativePath(key).split(CASE_KEY_SEPARATOR);
return id ?? null;
}
/** Exact Bun name pattern for a set of case ids (labels where the test name differs). */
export function caseTestNamePattern(ids: string[]): string {
const escaped = ids.map(id => (CASE_TEST_NAMES[id] ?? id).replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
return `(?:^|\\s)(?:${escaped.join('|')})$`;
}
/**
* Replace each case-sharded file with one key per registered case of `tier`.
* Throws when such a file's registration is not statically complete: an
* unregistered case would otherwise silently never run.
*/
export function expandCaseShards(files: string[], tier: PaidTier, rootDir = ROOT,
touchfiles: Record<string, string[]> = E2E_TOUCHFILES, tiers: Record<string, string> = E2E_TIERS): string[] {
return files.flatMap(file => {
const rel = normalizeRelativePath(file);
if (!CASE_SHARDED_FILES.includes(rel)) return [file];
const { registered, known } = fileCaseRegistration(rel, fs.readFileSync(path.join(rootDir, rel), 'utf8'), touchfiles, tiers);
if (!known) throw new Error(`Case-sharded ${rel} needs a complete literal case registration`);
return registered.filter(id => tiers[id] === tier).sort().map(id => `${rel}${CASE_KEY_SEPARATOR}${id}`);
});
}
/** Compatibility helper for callers that only need the effective wall. */
export function resolvePaidShardTimeoutMs(files: string[], explicitTimeoutMs?: number): number {
return resolvePaidShardBudget(files, explicitTimeoutMs).timeoutMs;
@@ -157,21 +219,47 @@ export interface TierClassification {
* guard runs and self-skips.
*/
export function classifyPaidTestFile(source: string, tier: PaidTier): TierClassification {
const other: PaidTier = tier === 'gate' ? 'periodic' : 'gate';
const declares = (candidate: PaidTier) =>
new RegExp(`EVALS_TIER\\s*===\\s*['"\`]${candidate}['"\`]`).test(source) ||
new RegExp(`\\b(?:describeE2ETier|e2eTierEnabled)\\(\\s*['"\`]${candidate}['"\`]`).test(source);
if (declares(tier)) return { included: true, reason: `declares tier '${tier}'` };
if (declares(other)) return { included: false, reason: `declares tier '${other}' only` };
const others = PAID_TIERS.filter(candidate => candidate !== tier && declares(candidate));
if (others.length) return { included: false, reason: `declares tier ${others.map(other => `'${other}'`).join(' and ')} only` };
return { included: true, reason: 'no whole-file tier guard — runtime E2E_TIERS filter decides' };
}
/**
* The E2E ids a paid file registers: the touchfile registrations that list the
* file. `known` is true only when those ids are complete: no computed
* registration (testName, *IfSelected, describeIfSelected with a non-literal
* argument) and every literal registration argument is among them. Quoted
* strings elsewhere (comments, skill paths) never count.
*/
export function fileCaseRegistration(
file: string, source: string,
touchfiles: Record<string, string[]> = E2E_TOUCHFILES,
tiers: Record<string, string> = E2E_TIERS,
): { registered: string[]; known: boolean } {
const rel = normalizeRelativePath(file);
const registered = Object.keys(touchfiles).filter(key => touchfiles[key]!.includes(rel));
const computed = /testName\s*:\s*(?!string\b)(?:`[^`]*\$\{|[A-Za-z_$])/.test(source)
|| /\btest(?:Concurrent)?IfSelected\s*\(\s*(?:`[^`]*\$\{|[A-Za-z_$])/.test(source)
|| /\bdescribeIfSelected\s*\([^,]*,(?!\s*\[)/.test(source)
|| [...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)].some(m => m[1]!.split(',')
.map(item => item.trim()).some(item => item && !/^(['"`])[^'"`$]*\1$/.test(item)));
const literal = [
...[...source.matchAll(/testName\s*:\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!),
...[...source.matchAll(/\btest(?:Concurrent)?IfSelected\s*\(\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!),
...[...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)]
.flatMap(m => [...m[1]!.matchAll(/(['"`])([^'"`]+)\1/g)].map(n => n[2]!)),
].filter(id => id in tiers);
return { registered, known: registered.length > 0 && !computed && literal.every(id => registered.includes(id)) };
}
/**
* A file is skipped for a tier lane only when its registered E2E ids are fully
* known and none of them has that tier. Ids are the touchfile registrations that
* list the file plus literal registration arguments (testName, *IfSelected);
* quoted strings elsewhere (comments, skill paths) never count. Any computed
* known (fileCaseRegistration) and none of them has that tier. Any computed
* registration, an id missing from the file's touchfile registration, or no id at
* all keeps today's scheduling (the child's runtime filter decides).
*/
@@ -180,26 +268,27 @@ export function tierSkipReason(
touchfiles: Record<string, string[]> = E2E_TOUCHFILES,
tiers: Record<string, string> = E2E_TIERS,
): string | null {
const rel = normalizeRelativePath(file);
const registered = Object.keys(touchfiles).filter(key => touchfiles[key]!.includes(rel));
if (!registered.length) return null;
const computed = /testName\s*:\s*(?:`[^`]*\$\{|[A-Za-z_$])/.test(source)
|| /\btest(?:Concurrent)?IfSelected\s*\(\s*(?:`[^`]*\$\{|[A-Za-z_$])/.test(source)
|| /\bdescribeIfSelected\s*\([^,]*,(?!\s*\[)/.test(source)
|| [...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)].some(m => m[1]!.split(',')
.map(item => item.trim()).some(item => item && !/^(['"`])[^'"`$]*\1$/.test(item)));
if (computed) return null;
const literal = [
...[...source.matchAll(/testName\s*:\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!),
...[...source.matchAll(/\btest(?:Concurrent)?IfSelected\s*\(\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!),
...[...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)]
.flatMap(m => [...m[1]!.matchAll(/(['"`])([^'"`]+)\1/g)].map(n => n[2]!)),
].filter(id => id in tiers);
if (literal.some(id => !registered.includes(id))) return null;
if (registered.some(id => tiers[id] === tier)) return null;
const { registered, known } = fileCaseRegistration(file, source, touchfiles, tiers);
if (!known || registered.some(id => tiers[id] === tier)) return null;
return `skipped: no E2E_TIERS id has tier ${tier}`;
}
/**
* The marathon lane selects positively: a file runs there only when it
* declares the marathon tier or registers a marathon-tier case. Files without
* marathon work never cost a marathon runner, and gate/periodic files never
* gain a third execution.
*/
export function marathonSkipReason(
file: string, source: string,
touchfiles: Record<string, string[]> = E2E_TOUCHFILES,
tiers: Record<string, string> = E2E_TIERS,
): string | null {
if (classifyPaidTestFile(source, 'marathon').reason === "declares tier 'marathon'") return null;
const { registered } = fileCaseRegistration(file, source, touchfiles, tiers);
return registered.some(id => tiers[id] === 'marathon') ? null : 'skipped: declares no marathon tier and registers no marathon case';
}
export interface TierSelection {
selected: string[];
excluded: Array<{ file: string; reason: string }>;
@@ -213,11 +302,11 @@ export function selectPaidTestFiles(files: string[], tier: PaidTier, rootDir = R
if (carveSkill && files.some(file => carveWrapper(file)) && !files.some(file => carveWrapper(file) === carveSkill)) {
throw new Error(`GSTACK_CARVE_SKILL=${carveSkill} has no generic section-loading wrapper`);
}
// Periodic-lane exclusions (documented-red / manual-hardware files): a
// Scheduled-lane exclusions (documented-red / manual-hardware files): a
// known-red weekly shard is triage waste locally AND in CI, so the list
// applies to every periodic run, with the reason surfaced per file.
// applies to every periodic and marathon run, with the reason surfaced per file.
const ciExcluded = (file: string): { reason: string; tracking: string } | undefined =>
tier === 'periodic' ? PERIODIC_CI_EXCLUDE[normalizeRelativePath(file)] : undefined;
tier !== 'gate' ? PERIODIC_CI_EXCLUDE[normalizeRelativePath(file)] : undefined;
for (const file of files) {
// One wrapper per process means a child-side return now creates an empty
// shard. Apply the existing explicit cost scope before planning processes.
@@ -233,7 +322,8 @@ export function selectPaidTestFiles(files: string[], tier: PaidTier, rootDir = R
}
const source = fs.readFileSync(path.join(rootDir, file), 'utf8');
const classification = classifyPaidTestFile(source, tier);
const skip = classification.included ? tierSkipReason(file, source, tier) : null;
const skip = !classification.included ? null
: tier === 'marathon' ? marathonSkipReason(file, source) : tierSkipReason(file, source, tier);
if (classification.included && !skip) selected.push(file);
else excluded.push({ file, reason: skip ?? classification.reason });
}
@@ -378,28 +468,29 @@ function packageVersionOnlySinceBase(rootDir: string, baseRef: string): boolean
}
/** Only audited per-case files, plus the separately selected judge, enter the fast profile. */
/** The selected PR-profile case ids a shard key owns (a case key owns at most its own case). */
function prProfileShardIds(key: string, selection: PaidCaseSelection): string[] {
const caseId = shardCaseId(key);
return (PR_PROFILE_FILES[shardFile(key)] ?? [])
.filter(id => (caseId === null || id === caseId) && (selection.e2e === null || selection.e2e.includes(id)));
}
export function prProfileFileSelected(file: string, selection: PaidCaseSelection): boolean {
if (file === 'test/skill-llm-eval.test.ts') return selection.judges === null || selection.judges.length > 0;
const ids = PR_PROFILE_FILES[normalizeRelativePath(file)];
return !!ids && (selection.e2e === null || ids.some(id => selection.e2e!.includes(id)));
return prProfileShardIds(file, selection).length > 0;
}
export function expectedPrCaseCount(file: string, selection: PaidCaseSelection): number {
if (file === 'test/skill-llm-eval.test.ts') return selection.judges?.length ?? Object.keys(LLM_JUDGE_TOUCHFILES).length;
return (PR_PROFILE_FILES[normalizeRelativePath(file)] ?? []).filter(id => selection.e2e === null || selection.e2e.includes(id)).length;
return prProfileShardIds(file, selection).length;
}
export function prProfileTestNamePattern(file: string, selection: PaidCaseSelection): string {
const labels: Record<string, string> = {
'plan-review-report': '/plan-eng-review writes GSTACK REVIEW REPORT to plan file',
'auq-format-gate': "/plan-ceo-review's first AskUserQuestion is a compliant decision brief (7/7 + substance)",
};
const ids = file === 'test/skill-llm-eval.test.ts'
? selection.judges ?? Object.keys(LLM_JUDGE_TOUCHFILES)
: (PR_PROFILE_FILES[file] ?? []).filter(id => selection.e2e === null || selection.e2e.includes(id));
: prProfileShardIds(file, selection);
if (ids.length === 0) throw new Error(`No selected PR cases for ${file}`);
const escaped = ids.map(id => (labels[id] ?? id).replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
return `(?:^|\\s)(?:${escaped.join('|')})$`;
return caseTestNamePattern(ids);
}
export function paidSelectionEnv(profile: PaidProfile, selection: PaidCaseSelection, reason: string): NodeJS.ProcessEnv {
@@ -443,6 +534,10 @@ export function diffSkipDecisionForFile(
options: DiffSkipOptions = {},
): ShardSkipDecision {
if (selectedNames === null) return { file, kept: true, reason: 'run-all selection' };
const caseId = shardCaseId(file);
if (caseId !== null) {
return selectedNames.has(caseId) ? { file, kept: true, reason: `selected: ${caseId}` } : { file, kept: false, reason: `case ${caseId} not selected` };
}
const rel = normalizeRelativePath(file);
if (!/^test\/skill-e2e-.*\.test\.ts$/.test(rel)) {
return { file, kept: true, reason: 'non-skill-e2e paid file — child self-skip authoritative' };
@@ -503,7 +598,7 @@ export function planPaidShards(
const shards: string[][] = [];
let pending: string[] = [];
for (const file of unique) {
if (isOverlayTestFile(file) || FILE_RETRY_BUDGETS.some(budget => budget.file === file)) {
if (isOverlayTestFile(file) || shardCaseId(file) !== null || FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file))) {
if (pending.length) shards.push(pending);
pending = [];
shards.push([file]);
@@ -524,7 +619,7 @@ export interface PaidShardBudget {
/** Explicit caller limits win; registered supervision preserves existing attempts. */
export function resolvePaidShardBudget(files: string[], overrideMs?: number): PaidShardBudget {
const finding = FILE_RETRY_BUDGETS.find(budget => files.map(normalizeRelativePath).includes(budget.file));
const finding = FILE_RETRY_BUDGETS.find(budget => files.map(shardFile).includes(budget.file));
if (finding && files.length !== 1) throw new Error('Registered retry budget requires its own shard');
if (overrideMs !== undefined && (!Number.isSafeInteger(overrideMs) || overrideMs <= 0 || overrideMs > 2_147_483_647)) {
throw new Error('Shard timeout must be a finite positive timer-safe integer');
@@ -534,8 +629,11 @@ export function resolvePaidShardBudget(files: string[], overrideMs?: number): Pa
if (overlay && overrideMs !== undefined && overrideMs < OVERLAY_MIN_FILE_WALL_MS) {
throw new Error(`Overlay shard requires at least ${OVERLAY_MIN_FILE_WALL_MS}ms; explicit wall ${overrideMs}ms cannot preserve its work and finalization budget`);
}
// A registered file's case shard supervises one case and its allowed attempts.
const registeredMs = finding && shardCaseId(files[0]!) !== null
? finding.caseMs * (finding.retries + 1) + finding.shardReserveMs : finding?.shardMs;
return {
timeoutMs: overrideMs ?? (finding ? finding.shardMs : overlay ? OVERLAY_MIN_FILE_WALL_MS : DEFAULT_SHARD_TIMEOUT_MS),
timeoutMs: overrideMs ?? (registeredMs ?? (overlay ? OVERLAY_MIN_FILE_WALL_MS : DEFAULT_SHARD_TIMEOUT_MS)),
source: overrideMs !== undefined ? 'explicit' : finding ? 'registered' : 'default',
policyId: finding?.id ?? null,
};
@@ -554,8 +652,8 @@ export function buildPaidShardArgs(
// Explicit --concurrent/--max-concurrency: the legacy path always set one;
// omitting it here made within-shard parallelism differ silently between
// the two runners (observed: 1.6x sumdur/wall sharded vs 8x legacy).
// Retries default to 1; RETRY_OVERRIDES membership (old matrix rows'
// earned `retries: 2`) flows through retriesForFiles at the call site.
// Retries come from retriesForFiles (the timeout-is-a-verdict rule) at the
// call site; the fallback of 1 serves only direct callers.
return ['test', ...files, '--retry', String(retries ?? 1), '--concurrent', `--max-concurrency=${maxConcurrency}`, `--timeout=${timeoutMs}`];
}
@@ -565,7 +663,8 @@ export function buildPaidShardArgs(
*/
export function shardSlug(files: string[]): string {
return files
.map((file) => path.basename(normalizeRelativePath(file)).replace(/\.test\.(?:[cm]?[jt]s|tsx|jsx)$/, ''))
.map((file) => path.basename(shardFile(file)).replace(/\.test\.(?:[cm]?[jt]s|tsx|jsx)$/, '')
+ (shardCaseId(file) === null ? '' : `--${shardCaseId(file)}`))
.join('+')
.replace(/[^a-zA-Z0-9._+-]/g, '-');
}
@@ -599,6 +698,8 @@ export interface ShardOutcome {
skippedTests: number | null;
/** Effective supervised wall; absent only for unstarted or legacy outcomes. */
budget?: PaidShardBudget;
/** Present when a verified receipt replaced execution (PR lane only). */
reused?: { inputKey: string; runId: string; revision: string; completedAt: number };
}
/**
@@ -655,6 +756,10 @@ export interface RunShardsOptions {
/** Fast-profile census: selected real cases per file, excluding Bun skips. */
expectedCases?: Record<string, number>;
casePatterns?: Record<string, string>;
/** The selected case ids per shard key (reported for reused shards). */
expectedCaseIds?: Record<string, string[]>;
/** PR lane only: verified reuse for one shard's exact child environment and wall. */
reuseFor?: (files: string[], env: NodeJS.ProcessEnv, budget: PaidShardBudget) => E2EShardReuse | null;
}
let shardLogSequence = 0;
@@ -703,17 +808,10 @@ export async function runPaidShard(
const log = options.log ?? ((line: string) => console.log(line));
const label = `[test:paid] shard ${shardNumber}/${totalShards}`;
const { command, args } = options.commandFor
? options.commandFor(files)
: {
command: process.execPath,
args: [...buildPaidShardArgs(
exactTestFileSelectors(files, rootDir),
timeoutMs,
options.withinShardConcurrency ?? DEFAULT_WITHIN_SHARD_CONCURRENCY,
retriesForFiles(files),
), ...(options.casePatterns ? ['--test-name-pattern', options.casePatterns[files[0]]] : [])],
};
// A case shard runs exactly its one case; PR patterns narrow further.
const caseId = files.length === 1 ? shardCaseId(files[0]!) : null;
const casePattern = options.casePatterns?.[files[0]!] ?? (caseId !== null ? caseTestNamePattern([caseId]) : undefined);
const expectedCases = options.expectedCases ?? (caseId !== null ? { [files[0]!]: 1 } : undefined);
const env = { ...(options.env ?? process.env) };
if (options.evalDirBase) {
@@ -725,6 +823,42 @@ export async function runPaidShard(
if (!env.GSTACK_CLAUDE_CLI_VERSION) {
env.GSTACK_CLAUDE_CLI_VERSION = getClaudeCliVersion();
}
// Verified first-attempt reuse (PR lane only; scripts/e2e-shard-reuse.ts):
// identical consumed inputs to a fresh pass in this PR replace execution
// with an explicitly reported reused result.
// Bootstrap-retention qualification binds per-run state, so that shard stays fresh.
const reuse = files.some(file => normalizeRelativePath(file) === 'test/skill-e2e-qa-workflow.test.ts')
? null : options.reuseFor?.(files, env, budget) ?? null;
const reused = reuse?.lookup() ?? null;
if (reused) {
const reusedFrom = { input_key: reused.key, run_id: reused.source.runId, revision: reused.source.revision,
completed_at: new Date(reused.source.completedAt).toISOString() };
const caseIds = options.expectedCaseIds?.[files[0]!] ?? [];
if (env.GSTACK_EVAL_DIR) {
fs.mkdirSync(env.GSTACK_EVAL_DIR, { recursive: true });
fs.writeFileSync(path.join(env.GSTACK_EVAL_DIR, `e2e-reused-${shardSlug(files)}.json`), `${JSON.stringify({
schema_version: 1, tier: 'e2e', shard: shardSlug(files), total_tests: caseIds.length, executed_tests: 0,
reused_tests: caseIds.length, passed: caseIds.length, failed: 0, total_cost_usd: 0, total_duration_ms: 0,
tests: caseIds.map(name => ({ name, suite: shardSlug(files), tier: 'e2e', passed: true, duration_ms: 0, cost_usd: 0,
execution: 'reused', reused_from: reusedFrom, attempt: 1 })),
}, null, 2)}\n`);
}
log(`${label} REUSED ${files.join(' ')} — identical inputs passed in run ${reused.source.runId} at ${reusedFrom.completed_at}`);
return { shard: shardNumber, files, status: 'passed', exitCode: 0, elapsedMs: 0, groupPid: null,
executedTests: caseIds.length, skippedTests: 0, budget,
reused: { inputKey: reused.key, runId: reused.source.runId, revision: reused.source.revision, completedAt: reused.source.completedAt } };
}
const { command, args } = options.commandFor
? options.commandFor(files)
: {
command: process.execPath,
args: [...buildPaidShardArgs(
exactTestFileSelectors(files.map(shardFile), rootDir),
timeoutMs,
options.withinShardConcurrency ?? DEFAULT_WITHIN_SHARD_CONCURRENCY,
retriesForFiles(files),
), ...(casePattern !== undefined ? ['--test-name-pattern', casePattern] : [])],
};
// Per-shard temp + Chromium-profile isolation — the free runner treats
// this as mandatory (test-free-shards.ts: two concurrent shards on one
// profile dir kill each other's browser; shared tmp cross-contaminates),
@@ -881,15 +1015,16 @@ export async function runPaidShard(
let status: ShardStatus = timedOut
? 'timed-out'
: !retentionFailed && !logWriteFailed && !incompleteCapture && strictTestExitCode(exitCode ?? 1, summary, expectedFiles) === 0 ? 'passed' : 'failed';
if (status === 'passed' && options.expectedCases) {
const expected = files.reduce((count, file) => count + (options.expectedCases![file] ?? 0), 0);
if (status === 'passed' && expectedCases) {
const expected = files.reduce((count, file) => count + (expectedCases[file] ?? 0), 0);
const actual = summary.terminalTestCounts.reduce((count, value) => count + value, 0) - summary.skippedTests;
if (expected < 1 || actual !== expected) {
status = 'failed';
log(`${label} expected ${expected} selected cases, executed ${actual}; refusing incomplete PR coverage`);
log(`${label} expected ${expected} selected cases, executed ${actual}; refusing incomplete case coverage`);
}
}
const elapsedMs = Date.now() - startedAt;
if (status === 'passed' && reuse) reuse.publish();
// Failure debuggability without the RAM cost: read back only the log's
// tail. Live mode already streamed everything, so no re-print there.
@@ -1077,6 +1212,16 @@ export interface ManifestEntry {
reason?: string;
/** Required when a registered retry-budget file is planned. */
budget?: PaidShardBudget;
/** Budget-mode packing weight (recorded wall, or the whole budget when unknown). */
estimatedMs?: number;
}
/** Budget-mode plan: per-executor estimate and the CI job timeout it needs. */
export interface PaidSlicePlan {
sliceBudgetMs: number;
jobs: number;
estimatedSliceMs: number[];
ciTimeoutMinutes: number;
}
export interface PaidRunManifest {
@@ -1089,43 +1234,57 @@ export interface PaidRunManifest {
profile?: PaidProfile;
selection?: PaidCaseSelection;
prCoverage?: PrProfileSelection;
plan?: PaidSlicePlan;
entries: ManifestEntry[];
}
/**
* Files whose old evals.yml matrix rows carried `retries: 2`, with the
* receipts that earned them (see the deleted rows' comments). The runner
* default stays --retry 1; membership here is a literals map so retry
* parity with the matrix is explicit, not folklore.
* Automatic retries follow the approved rule in test/helpers/eval-budgets.ts:
* a timed-out attempt is a verdict, so only files whose every case budget is
* at most RETRY_MAX_CASE_MS keep a retry (registered rows derive it from their
* caseMs; SHORT_CASE_RETRY_FILES lists the rest). Overlays and every other
* paid file run once. A multi-file shard takes the smallest allowance.
*/
export const RETRY_OVERRIDES: Record<string, number> = {
'test/skill-e2e-workflow.test.ts': 2,
'test/skill-e2e-office-hours-auto-mode.test.ts': 2,
'test/skill-e2e-plan-mode-no-op.test.ts': 2,
};
export function retriesForFiles(files: string[]): number {
if (files.some(isOverlayTestFile)) return 0;
return Math.max(1, ...files.map((f) => RETRY_OVERRIDES[normalizeRelativePath(f)] ?? 1));
return Math.min(...files.map((file) => {
const rel = shardFile(file);
const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === rel);
if (registered) return registered.retries;
return SHORT_CASE_RETRY_FILES.includes(rel) ? 1 : 0;
}));
}
export const PAID_TEST_DURATIONS_FILE = 'scripts/paid-test-durations.json';
/**
* Recorded per-file paid-shard wall times (ms) from real CI slice reports,
* refreshed with `--report <dir> --write-durations`. A packing hint only: a
* missing or corrupt seed keeps the supervision-budget allocation.
* Recorded per-file paid-shard wall times (ms) per tier from real CI slice
* reports (a file's gate and periodic cases differ), refreshed with
* `--report <dir> --write-durations`. A packing hint only: a missing or
* corrupt seed keeps the supervision-budget allocation.
*/
export function loadPaidTestDurations(rootDir = ROOT): Record<string, number> {
export function loadPaidTestDurations(rootDir = ROOT, tier: PaidTier = 'gate'): Record<string, number> {
try {
const parsed = JSON.parse(fs.readFileSync(path.join(rootDir, PAID_TEST_DURATIONS_FILE), 'utf8')) as { durations?: Record<string, unknown> };
return Object.fromEntries(Object.entries(parsed.durations ?? {})
const parsed = JSON.parse(fs.readFileSync(path.join(rootDir, PAID_TEST_DURATIONS_FILE), 'utf8')) as { version?: unknown; tiers?: Record<string, Record<string, unknown>> };
if (parsed.version !== 2) return {};
return Object.fromEntries(Object.entries(parsed.tiers?.[tier] ?? {})
.filter((entry): entry is [string, number] => typeof entry[1] === 'number' && Number.isFinite(entry[1]) && entry[1] > 0));
} catch {
return {};
}
}
/** Rewrite one tier of the committed seed, keeping the other tiers. */
export function writePaidTestDurations(tier: PaidTier, durations: Record<string, number>, rootDir = ROOT): void {
const target = path.join(rootDir, PAID_TEST_DURATIONS_FILE);
const tiers = Object.fromEntries(PAID_TIERS.map(name => [name, loadPaidTestDurations(rootDir, name)])
.filter(([name, recorded]) => name === tier || Object.keys(recorded as object).length > 0));
tiers[tier] = durations;
const temporary = `${target}.tmp-${process.pid}`;
fs.writeFileSync(temporary, `${JSON.stringify({ version: 2, recordedAt: new Date().toISOString(), tiers }, null, 2)}\n`);
fs.renameSync(temporary, target);
}
/** Merge a report's executed single-file outcomes into the seed; all-skipped shards carry no cost signal. */
export function mergePaidTestDurations(seed: Record<string, number>, results: SliceResult[]): Record<string, number> {
const merged = { ...seed };
@@ -1138,6 +1297,65 @@ export function mergePaidTestDurations(seed: Record<string, number>, results: Sl
return Object.fromEntries(Object.entries(merged).sort(([a], [b]) => (a < b ? -1 : 1)));
}
/** Setup, image pull and artifact upload allowance on top of a slice's supervised wall. */
export const CI_SETUP_ALLOWANCE_MINUTES = 20;
/** Estimated wall of one executor running `files` in order on `jobs` FIFO workers. */
export function estimatedSliceMs(files: string[], weight: (file: string) => number, jobs: number): number {
const workers = Array<number>(Math.max(1, jobs)).fill(0);
for (const file of files) {
const next = workers.indexOf(Math.min(...workers));
workers[next] += weight(file);
}
return Math.max(...workers);
}
/** Executor order within a slice: longest recorded work first, then by path. */
export function sliceExecutionOrder<T extends { file: string; estimatedMs?: number }>(entries: T[]): T[] {
return [...entries].sort((a, b) => (b.estimatedMs ?? 0) - (a.estimatedMs ?? 0) || (a.file < b.file ? -1 : a.file > b.file ? 1 : 0));
}
/** Supervised worst case of one slice in execution order, overlays at their own admission limit. */
export function sliceSupervisedWallMs(files: string[], jobs: number, overrideMs?: number): number {
return paidShardWallUpperBoundMs(files.filter(file => !isOverlayTestFile(file)), jobs, overrideMs)
+ paidShardWallUpperBoundMs(files.filter(isOverlayTestFile), Math.min(jobs, OVERLAY_MAX_ACTIVE_SHARDS), overrideMs);
}
/**
* Budget packing: one runner per file or per tightly packed group. Files go
* longest-recorded-first into the fullest slice whose estimated wall stays
* within the budget (best fit), else into a new slice. A file with no recorded
* wall weighs the whole budget, so unknown cost gets a runner of its own. A
* file longer than the budget runs alone. Overlay wrappers keep one shared
* final slice (one wrapper at a time). The CI timeout covers every slice's
* supervised worst case plus the setup allowance.
*/
export function packBySliceBudget(files: string[], budgetMs: number, jobs: number,
recorded: Record<string, number>, timeoutMs?: number): {
slices: string[][]; estimates: Record<string, number>; estimatedSliceMs: number[]; ciTimeoutMinutes: number;
} {
const estimates = Object.fromEntries(files.map(file => [file, recorded[normalizeRelativePath(file)] ?? budgetMs]));
const weight = (file: string) => estimates[file]!;
const slices: string[][] = [];
for (const file of sliceExecutionOrder(files.filter(file => !isOverlayTestFile(file)).map(file => ({ file, estimatedMs: weight(file) }))).map(entry => entry.file)) {
let best = -1, bestMs = -1;
slices.forEach((planned, index) => {
const ms = estimatedSliceMs([...planned, file], weight, jobs);
if (ms <= budgetMs && ms > bestMs) { best = index; bestMs = ms; }
});
if (best < 0) slices.push([file]);
else slices[best]!.push(file);
}
const overlays = files.filter(isOverlayTestFile).sort();
if (overlays.length) slices.push(sliceExecutionOrder(overlays.map(file => ({ file, estimatedMs: weight(file) }))).map(entry => entry.file));
if (!slices.length) slices.push([]);
const estimatedSliceMsList = slices.map(planned => planned.some(isOverlayTestFile)
? estimatedSliceMs(planned, weight, Math.min(jobs, OVERLAY_MAX_ACTIVE_SHARDS)) : estimatedSliceMs(planned, weight, jobs));
const worst = Math.max(0, ...slices.map(planned => sliceSupervisedWallMs(planned, jobs, timeoutMs)));
return { slices, estimates, estimatedSliceMs: estimatedSliceMsList,
ciTimeoutMinutes: Math.ceil(worst / 60_000) + CI_SETUP_ALLOWANCE_MINUTES };
}
/** Worker counts whose worst-case slice wall duration packing may never worsen. */
export const SUPERVISED_WORKER_COUNTS = [1, 2, 3, 4] as const;
@@ -1153,7 +1371,13 @@ export const SUPERVISED_WORKER_COUNTS = [1, 2, 3, 4] as const;
export function buildRunManifest(opts: {
tier: PaidTier;
profile?: PaidProfile;
sliceCount: number;
/** Fixed slice count; exclusive with sliceBudgetMs. */
sliceCount?: number;
/** Budget mode: pack recorded work so each executor's estimated wall stays
* within this budget; the slice count follows from the plan. */
sliceBudgetMs?: number;
/** Shard workers per executor (EVALS_JOBS) that budget mode plans for. */
jobs?: number;
evalsAll: boolean;
timeoutMs?: number;
discovered?: string[];
@@ -1165,7 +1389,12 @@ export function buildRunManifest(opts: {
/** Weekly gate census only: LLM judges already run in the periodic census and PR gate lanes. */
skipJudges?: boolean;
}): PaidRunManifest {
if (!Number.isInteger(opts.sliceCount) || opts.sliceCount <= 0) {
const budgetMode = opts.sliceBudgetMs !== undefined;
if (budgetMode === (opts.sliceCount !== undefined)) throw new Error('Plan with exactly one of --slices or --slice-budget');
if (budgetMode && (!Number.isSafeInteger(opts.sliceBudgetMs) || opts.sliceBudgetMs! <= 0 || !Number.isSafeInteger(opts.jobs) || opts.jobs! <= 0)) {
throw new Error('--slice-budget needs a positive budget and an explicit positive --jobs');
}
if (!budgetMode && (!Number.isInteger(opts.sliceCount) || opts.sliceCount! <= 0)) {
throw new Error(`--slices needs a positive integer. Received: ${opts.sliceCount}`);
}
const rootDir = opts.rootDir ?? ROOT;
@@ -1178,7 +1407,7 @@ export function buildRunManifest(opts: {
const selected = opts.skipJudges ? tierSelection.selected.filter(file => !judge(file)) : tierSelection.selected;
const excluded = [...tierSelection.excluded, ...(opts.skipJudges ? tierSelection.selected.filter(judge)
.map(file => ({ file, reason: 'skipped: LLM judges run in the periodic census and PR gate lanes' })) : [])];
const shards = planPaidShards(selected, { maxFilesPerShard: 1 });
const shards = planPaidShards(expandCaseShards(selected, opts.tier, rootDir), { maxFilesPerShard: 1 });
const cases = computePaidCaseSelection({ profile, env, rootDir, changedFiles: opts.changedFiles });
const fast = cases.coverage?.mode === 'pr';
const profileShards = fast ? shards.filter(files => prProfileFileSelected(files[0], cases.selection)) : shards;
@@ -1189,14 +1418,32 @@ export function buildRunManifest(opts: {
}
const entries: ManifestEntry[] = [];
const overlaySlice = opts.sliceCount;
if (budgetMode) {
const plan = packBySliceBudget(runnable.map(files => files[0]!), opts.sliceBudgetMs!, opts.jobs!,
opts.durations ?? loadPaidTestDurations(rootDir, opts.tier), opts.timeoutMs);
plan.slices.forEach((files, index) => files.forEach(file => entries.push({ file, slice: index + 1, status: 'planned',
estimatedMs: plan.estimates[file]!,
...(FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file)) ? { budget: resolvePaidShardBudget([file], opts.timeoutMs) } : {}) })));
for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason });
for (const e of excluded) entries.push({ file: e.file, slice: 0, status: 'excluded', reason: e.reason });
entries.sort((a, b) => (a.file < b.file ? -1 : 1));
return parseRunManifest(JSON.stringify({
version: 1, tier: opts.tier, evalsAll: opts.evalsAll, sliceCount: plan.slices.length,
selectionReason: cases.reason, profile, selection: cases.selection,
...(cases.coverage ? { prCoverage: cases.coverage } : {}),
plan: { sliceBudgetMs: opts.sliceBudgetMs!, jobs: opts.jobs!, estimatedSliceMs: plan.estimatedSliceMs, ciTimeoutMinutes: plan.ciTimeoutMinutes },
entries,
} satisfies PaidRunManifest));
}
const sliceCount = opts.sliceCount!;
const overlaySlice = sliceCount;
const reserveOverlaySlice = overlaySlice > 1 && runnable.some(files => files.some(isOverlayTestFile));
const ordinarySlices = overlaySlice - Number(reserveOverlaySlice);
// Spread registered long files by supervised load. Keep one ordinary-only
// lane when possible, so every lane does not inherit a long-workflow tail.
// The reserved overlay slice retains its ownership.
const ordinary = runnable.filter(files => !files.some(isOverlayTestFile));
const registered = ordinary.filter(files => FILE_RETRY_BUDGETS.some(budget => budget.file === files[0]));
const registered = ordinary.filter(files => FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(files[0]!)));
const allocations = new Map<string, number>();
if (registered.length && ordinarySlices > 1) {
const loads = Array<number>(ordinarySlices).fill(0);
@@ -1220,7 +1467,7 @@ export function buildRunManifest(opts: {
}
const packed = packByRecordedDuration();
function packByRecordedDuration(): Map<string, number> | null {
const recorded = opts.durations ?? loadPaidTestDurations(rootDir);
const recorded = opts.durations ?? loadPaidTestDurations(rootDir, opts.tier);
if (ordinarySlices < 2 || ordinary.length === 0 || Object.keys(recorded).length === 0) return null;
const bound = (files: string[], jobs: number) => paidShardWallUpperBoundMs([...files].sort(), jobs, opts.timeoutMs);
const lanes = Array.from({ length: ordinarySlices }, (_, lane) =>
@@ -1267,7 +1514,7 @@ export function buildRunManifest(opts: {
runnable.forEach((files) => {
const slice = files.some(isOverlayTestFile) ? overlaySlice : (packed ?? allocations).get(files[0])!;
entries.push({ file: files[0], slice, status: 'planned',
...(FILE_RETRY_BUDGETS.some(budget => budget.file === files[0])
...(FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(files[0]!))
? { budget: resolvePaidShardBudget(files, opts.timeoutMs) } : {}) });
});
for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason });
@@ -1278,7 +1525,7 @@ export function buildRunManifest(opts: {
version: 1,
tier: opts.tier,
evalsAll: opts.evalsAll,
sliceCount: opts.sliceCount,
sliceCount,
selectionReason: cases.reason,
profile,
selection: cases.selection,
@@ -1288,10 +1535,25 @@ export function buildRunManifest(opts: {
return parseRunManifest(JSON.stringify(manifest));
}
/**
* Narrow a built manifest to a curated case subset (validation phases): the
* selection binds every child, and a case shard outside it can execute
* nothing, so it becomes skipped instead of an empty planned shard.
*/
export function restrictManifestSelection(manifest: PaidRunManifest, selection: PaidCaseSelection, reason: string): PaidRunManifest {
const entries = manifest.entries.map(entry => {
const caseId = shardCaseId(entry.file);
if (entry.status !== 'planned' || caseId === null || selection.e2e === null || selection.e2e.includes(caseId)) return entry;
const { estimatedMs: _estimate, budget: _budget, ...rest } = entry;
return { ...rest, slice: 0, status: 'skipped-by-diff' as const, reason };
});
return parseRunManifest(JSON.stringify({ ...manifest, selection, entries }));
}
export function parseRunManifest(raw: string): PaidRunManifest {
const parsed = JSON.parse(raw) as PaidRunManifest;
if (parsed.version !== 1) throw new Error(`unsupported manifest version: ${(parsed as { version?: unknown }).version}`);
if (parsed.tier !== 'gate' && parsed.tier !== 'periodic') throw new Error(`manifest tier invalid: ${parsed.tier}`);
if (!PAID_TIERS.includes(parsed.tier)) throw new Error(`manifest tier invalid: ${parsed.tier}`);
if (parsed.profile !== undefined && parsed.profile !== 'pr' && parsed.profile !== 'full') throw new Error('manifest profile invalid');
if (parsed.selection !== undefined) {
for (const [key, inventory] of [['e2e', E2E_TOUCHFILES], ['judges', LLM_JUDGE_TOUCHFILES]] as const) {
@@ -1329,19 +1591,50 @@ export function parseRunManifest(raw: string): PaidRunManifest {
if (entry.status === 'planned' && (entry.slice < 1 || entry.slice > parsed.sliceCount)) {
throw new Error(`planned entry ${entry.file} has out-of-range slice ${entry.slice}`);
}
if (entry.estimatedMs !== undefined && (entry.status !== 'planned' || !Number.isSafeInteger(entry.estimatedMs) || entry.estimatedMs < 0)) {
throw new Error(`manifest entry ${entry.file} has an invalid estimate`);
}
if (entry.status === 'planned' && parsed.prCoverage?.mode === 'pr' && !prProfileFileSelected(entry.file, parsed.selection!)) {
throw new Error(`manifest file is outside its PR case selection: ${entry.file}`);
}
}
for (const entry of parsed.entries) {
const caseId = shardCaseId(entry.file);
if (caseId === null ? entry.status === 'planned' && CASE_SHARDED_FILES.includes(shardFile(entry.file))
: !CASE_SHARDED_FILES.includes(shardFile(entry.file)) || !(caseId in E2E_TOUCHFILES)) {
throw new Error(`Case-sharded files plan one registered case per shard: ${entry.file}`);
}
if (caseId !== null && entry.status === 'planned' && parsed.selection?.e2e && !parsed.selection.e2e.includes(caseId)) {
throw new Error(`Planned case shard is outside the manifest selection: ${entry.file}`);
}
}
if (parsed.prCoverage?.mode === 'pr') {
const required = Object.entries(PR_PROFILE_FILES).filter(([, ids]) => ids.some(id => parsed.selection!.e2e!.includes(id))).map(([file]) => file);
if (parsed.selection!.judges!.length) required.push('test/skill-llm-eval.test.ts');
for (const file of required) {
if (parsed.entries.filter(entry => entry.file === file && entry.status === 'planned').length !== 1) {
throw new Error(`PR selected cases require exactly one planned owning file: ${file}`);
const planned = parsed.entries.filter(entry => entry.status === 'planned').map(entry => normalizeRelativePath(entry.file));
const required: string[][] = Object.entries(PR_PROFILE_FILES).flatMap(([file, ids]) => {
const selected = ids.filter(id => parsed.selection!.e2e!.includes(id));
if (!selected.length) return [];
return [CASE_SHARDED_FILES.includes(file) ? selected.map(id => `${file}#${id}`) : [file]];
});
if (parsed.selection!.judges!.length) required.push(['test/skill-llm-eval.test.ts']);
for (const owners of required) {
if (owners.some(owner => planned.filter(key => key === owner).length !== 1)
|| planned.filter(key => shardFile(key) === shardFile(owners[0]!)).length !== owners.length) {
throw new Error(`PR selected cases require exactly one planned owning file: ${owners.join(', ')}`);
}
}
}
if (parsed.plan !== undefined) {
const plan = parsed.plan;
const count = (value: unknown) => Number.isSafeInteger(value) && Number(value) > 0;
if (!plan || typeof plan !== 'object' || !count(plan.sliceBudgetMs) || !count(plan.jobs) || !count(plan.ciTimeoutMinutes)
|| !Array.isArray(plan.estimatedSliceMs) || plan.estimatedSliceMs.length !== parsed.sliceCount
|| !plan.estimatedSliceMs.every(ms => Number.isSafeInteger(ms) && ms >= 0)
|| parsed.entries.some(entry => entry.status === 'planned' && entry.estimatedMs === undefined)) {
throw new Error('manifest slice plan malformed');
}
}
const keys = parsed.entries.map(entry => normalizeRelativePath(entry.file));
if (new Set(keys).size !== keys.length) throw new Error('Duplicate manifest entry');
const overlaySlice = parsed.sliceCount;
const plannedOverlays = parsed.entries.filter(entry => entry.status === 'planned' && isOverlayTestFile(entry.file));
if (plannedOverlays.some(entry => entry.slice !== overlaySlice)) {
@@ -1352,8 +1645,8 @@ export function parseRunManifest(raw: string): PaidRunManifest {
throw new Error('The final ordinary manifest slice is reserved for overlay files');
}
for (const budget of FILE_RETRY_BUDGETS) {
const entries = parsed.entries.filter(entry => normalizeRelativePath(entry.file) === budget.file);
if (entries.length > 1) throw new Error(`Duplicate registered manifest entry: ${budget.file}`);
const entries = parsed.entries.filter(entry => shardFile(entry.file) === budget.file);
if (entries.length > 1 && entries.some(entry => shardCaseId(entry.file) === null)) throw new Error(`Duplicate registered manifest entry: ${budget.file}`);
for (const entry of entries.filter(entry => entry.status === 'planned')) {
if (!entry.budget) throw new Error(`Registered manifest needs an explicit budget record: ${budget.file}`);
const expected = resolvePaidShardBudget([entry.file], entry.budget.source === 'explicit' ? entry.budget.timeoutMs : undefined);
@@ -1371,7 +1664,7 @@ export interface SliceResult {
sliceIndex: number;
sliceCount: number;
timeoutOverrideMs?: number;
outcomes: Array<Pick<ShardOutcome, 'files' | 'status' | 'exitCode' | 'elapsedMs' | 'executedTests' | 'skippedTests' | 'budget'>>;
outcomes: Array<Pick<ShardOutcome, 'files' | 'status' | 'exitCode' | 'elapsedMs' | 'executedTests' | 'skippedTests' | 'budget' | 'reused'>>;
}
/**
@@ -1405,7 +1698,7 @@ export function verifySliceResults(
const reported = new Map<string, { slice: number; status: ShardStatus }>();
for (const result of results) {
for (const outcome of result.outcomes) {
if (outcome.files.some(file => FILE_RETRY_BUDGETS.some(budget => budget.file === normalizeRelativePath(file))) && outcome.files.length !== 1) {
if (outcome.files.some(file => FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file)) || shardCaseId(file) !== null) && outcome.files.length !== 1) {
problems.push('Registered result must report its own shard');
}
const file = normalizeRelativePath(outcome.files[0] ?? '');
@@ -1417,9 +1710,21 @@ export function verifySliceResults(
problems.push(`PR profile expected ${expected} executed cases in ${file}, received ${executed}`);
}
}
if (shardCaseId(file) !== null && outcome.status === 'passed'
&& (outcome.executedTests === null || outcome.skippedTests === null || outcome.executedTests - outcome.skippedTests !== 1)) {
problems.push(`Case shard must execute exactly its one case: ${file}`);
}
if (outcome.reused !== undefined) {
const r = outcome.reused;
if (manifest.prCoverage?.mode !== 'pr') problems.push(`${file}: only the fast PR profile may reuse results; this lane executes fresh`);
if (outcome.status !== 'passed' || outcome.exitCode !== 0 || !/^[a-f0-9]{64}$/.test(r?.inputKey ?? '')
|| !/^[\w./-]{1,160}$/.test(r?.runId ?? '') || !/^[a-f0-9]{40}$/.test(r?.revision ?? '') || !Number.isSafeInteger(r?.completedAt) || r.completedAt <= 0) {
problems.push(`${file}: malformed reused result`);
}
}
if (reported.has(file)) problems.push(`${file} reported by two slices`);
reported.set(file, { slice: result.sliceIndex, status: outcome.status });
const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === file);
const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === shardFile(file));
const finding = STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === file);
if (finding) {
// Full-census runs must account for every registered case. A manifest
@@ -1453,6 +1758,22 @@ export function verifySliceResults(
return { ok: problems.length === 0, problems };
}
/** Budget-mode plan lines: every slice with its estimate, files and retries. */
export function formatSlicePlan(manifest: PaidRunManifest): string[] {
const plan = manifest.plan;
if (!plan) return [];
const minutes = (ms: number) => (ms / 60_000).toFixed(1);
const lines = [`[test:paid] slice plan: ${manifest.sliceCount} slice(s) x ${plan.jobs} worker(s), budget ${minutes(plan.sliceBudgetMs)}m per slice, `
+ `longest estimate ${minutes(Math.max(0, ...plan.estimatedSliceMs))}m, CI job timeout ${plan.ciTimeoutMinutes}m`];
for (let slice = 1; slice <= manifest.sliceCount; slice++) {
const mine = sliceExecutionOrder(manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice));
const over = plan.estimatedSliceMs[slice - 1]! > plan.sliceBudgetMs ? ' [over budget: longer than one runner allows]' : '';
lines.push(` slice ${slice}: ~${minutes(plan.estimatedSliceMs[slice - 1]!)}m${over}`);
for (const entry of mine) lines.push(` ${entry.file} ~${minutes(entry.estimatedMs ?? 0)}m retries=${retriesForFiles([entry.file])}`);
}
return lines;
}
export function formatProfileCoverage(manifest: PaidRunManifest): string[] {
const coverage = manifest.prCoverage;
return [
@@ -1497,6 +1818,9 @@ type CliOptions = {
emitPlanPath: string | null;
/** Slice count for --emit-plan. */
slices: number;
/** Budget mode for --emit-plan / --list: per-executor estimated wall. */
sliceBudgetMs: number | null;
jobsExplicit: boolean;
/** Executor mode: consume this manifest... */
planPath: string | null;
/** ...running only this 1-based slice. */
@@ -1519,10 +1843,10 @@ function validatedTier(value: string | undefined, source: string): PaidTier {
// otherwise cast through unchecked, match nothing in the runtime E2E_TIERS
// filter, self-skip every test, and exit 0 with all shards 'passed' — the
// exact 0%-execution-looks-like-a-pass class this runner exists to kill.
if (value !== 'gate' && value !== 'periodic') {
throw new Error(`${source} must be gate or periodic. Received: ${value}`);
if (!PAID_TIERS.includes(value as PaidTier)) {
throw new Error(`${source} must be gate, periodic or marathon. Received: ${value}`);
}
return value;
return value as PaidTier;
}
function validatedProfile(value: string | undefined, source: string): PaidProfile {
@@ -1553,6 +1877,8 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process
maxFilesPerShard: DEFAULT_MAX_FILES_PER_SHARD,
emitPlanPath: null,
slices: 1,
sliceBudgetMs: null,
jobsExplicit: !!env.EVALS_JOBS,
planPath: null,
sliceIndex: null,
reportDir: null,
@@ -1563,9 +1889,7 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process
const arg = argv[index];
if (arg === '--list') { options.listOnly = true; continue; }
if (arg === '--tier') {
const value = argv[index += 1];
if (value !== 'gate' && value !== 'periodic') throw new Error(`--tier must be gate or periodic. Received: ${value}`);
options.tier = value;
options.tier = validatedTier(argv[index += 1] ?? '-', '--tier');
continue;
}
if (arg === '--profile') {
@@ -1574,7 +1898,8 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process
options.profile = validatedProfile(value, '--profile'); options.profileExplicit = true; continue;
}
if (arg === '--timeout') { options.timeoutMs = parsePositiveInt(argv[index += 1], '--timeout') * 1000; options.timeoutExplicit = true; continue; }
if (arg === '--jobs') { options.jobs = parsePositiveInt(argv[index += 1], '--jobs'); continue; }
if (arg === '--jobs') { options.jobs = parsePositiveInt(argv[index += 1], '--jobs'); options.jobsExplicit = true; continue; }
if (arg === '--slice-budget') { options.sliceBudgetMs = parsePositiveInt(argv[index += 1], '--slice-budget') * 1000; continue; }
if (arg === '--files-per-shard') { options.maxFilesPerShard = parsePositiveInt(argv[index += 1], '--files-per-shard'); continue; }
if (arg === '--emit-plan') {
const value = argv[index += 1];
@@ -1598,6 +1923,8 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process
throw new Error(`Unknown argument: ${arg}`);
}
if (options.writeDurations && !options.reportDir) throw new Error('--write-durations requires --report');
if (options.sliceBudgetMs !== null && argv.includes('--slices')) throw new Error('Plan with exactly one of --slices or --slice-budget');
if (options.sliceBudgetMs !== null && !options.jobsExplicit) throw new Error('--slice-budget needs explicit --jobs (or EVALS_JOBS): the plan packs and supervises for that worker count');
if (options.skipJudges && (!options.emitPlanPath || options.tier !== 'gate')) throw new Error('--skip-judges applies only to an emitted gate census plan');
if (options.profile === 'pr' && options.tier !== 'gate') throw new Error('PR profile requires gate tier');
if (options.profile === 'pr' && options.maxFilesPerShard !== 1) throw new Error('PR profile requires one file per shard to preserve case accounting');
@@ -1613,7 +1940,7 @@ async function main(): Promise<number> {
const manifest = buildRunManifest({
tier: options.tier,
profile: options.profile,
sliceCount: options.slices,
...(options.sliceBudgetMs !== null ? { sliceBudgetMs: options.sliceBudgetMs, jobs: options.jobs } : { sliceCount: options.slices }),
timeoutMs: options.timeoutExplicit ? options.timeoutMs : undefined,
evalsAll: process.env.EVALS_ALL === '1',
skipJudges: options.skipJudges,
@@ -1628,6 +1955,7 @@ async function main(): Promise<number> {
+ `${planned} planned across ${manifest.sliceCount} slice(s), ${skipped} skipped by diff, `
+ `${excludedCount} excluded (${manifest.selectionReason})`,
);
for (const line of formatSlicePlan(manifest)) console.log(line);
return 0;
}
@@ -1647,16 +1975,14 @@ async function main(): Promise<number> {
for (const line of formatProfileCoverage(manifest)) console.log(line);
for (const result of results.sort((a, b) => a.sliceIndex - b.sliceIndex)) {
for (const outcome of result.outcomes) {
console.log(` slice ${result.sliceIndex} ${outcome.status.padEnd(15)} ${String(Math.round(outcome.elapsedMs / 1000)).padStart(5)}s ${outcome.files.join(' ')}`);
const shown = outcome.reused ? `reused (run ${outcome.reused.runId})` : outcome.status;
console.log(` slice ${result.sliceIndex} ${shown.padEnd(15)} ${String(Math.round(outcome.elapsedMs / 1000)).padStart(5)}s ${outcome.files.join(' ')}`);
}
}
if (options.writeDurations) {
const durations = mergePaidTestDurations(loadPaidTestDurations(), results);
const target = path.join(ROOT, PAID_TEST_DURATIONS_FILE);
const temporary = `${target}.tmp-${process.pid}`;
fs.writeFileSync(temporary, `${JSON.stringify({ version: 1, recordedAt: new Date().toISOString(), durations }, null, 2)}\n`);
fs.renameSync(temporary, target);
console.log(`[test:paid] wrote ${Object.keys(durations).length} durations to ${PAID_TEST_DURATIONS_FILE}`);
const durations = mergePaidTestDurations(loadPaidTestDurations(ROOT, manifest.tier), results);
writePaidTestDurations(manifest.tier, durations);
console.log(`[test:paid] wrote ${Object.keys(durations).length} ${manifest.tier} durations to ${PAID_TEST_DURATIONS_FILE}`);
}
// Historical flaky_retries includes every case with multiple attempts,
// whether its final result passed or failed. Report attempts separately
@@ -1767,7 +2093,10 @@ async function main(): Promise<number> {
if (options.sliceIndex > manifest.sliceCount) {
throw new Error(`--slice ${options.sliceIndex} exceeds manifest sliceCount ${manifest.sliceCount}`);
}
const mine = manifest.entries.filter((e) => e.status === 'planned' && e.slice === options.sliceIndex);
if (manifest.plan && options.jobs !== manifest.plan.jobs) {
throw new Error(`manifest was packed for ${manifest.plan.jobs} worker(s) per slice; EVALS_JOBS=${options.jobs} would break its supervision bound`);
}
const mine = sliceExecutionOrder(manifest.entries.filter((e) => e.status === 'planned' && e.slice === options.sliceIndex));
const shards = mine.map((e) => [e.file]);
for (const files of shards) resolvePaidShardTimeoutMs(files, timeoutOverride);
console.log(`[test:paid] slice ${options.sliceIndex}/${manifest.sliceCount}: ${shards.length} shard(s), tier=${manifest.tier}, evalsAll=${manifest.evalsAll}`);
@@ -1775,7 +2104,7 @@ async function main(): Promise<number> {
if (options.listOnly) {
for (const [index, files] of shards.entries()) {
const budget = resolvePaidShardBudget(files, timeoutOverride);
console.log(` shard ${index + 1}/${shards.length}: ${files.join(' ')} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'}`);
console.log(` shard ${index + 1}/${shards.length}: ${files.join(' ')} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'} retries=${retriesForFiles(files)}`);
}
return 0;
}
@@ -1794,6 +2123,18 @@ async function main(): Promise<number> {
...(manifest.prCoverage?.mode === 'pr' ? {
expectedCases: Object.fromEntries(mine.map(entry => [entry.file, expectedPrCaseCount(entry.file, manifest.selection!)])),
casePatterns: Object.fromEntries(mine.map(entry => [entry.file, prProfileTestNamePattern(entry.file, manifest.selection!)])),
expectedCaseIds: Object.fromEntries(mine.map(entry => [entry.file, prProfileShardIds(entry.file, manifest.selection!)])),
reuseFor: e2eReuseLaneProblem(process.env, manifest.prCoverage.mode) !== null ? undefined : (files, env, budget) => {
const key = files[0]!;
const file = shardFile(key);
if (files.length !== 1 || !/^test\/skill-e2e-/.test(file)) return null;
const { registered, known } = fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8'));
return prepareE2EShardReuse({ root: ROOT, key, file, caseIds: prProfileShardIds(key, manifest.selection!),
registeredIds: registered, registrationKnown: known,
casePattern: prProfileTestNamePattern(key, manifest.selection!), expectedCases: expectedPrCaseCount(key, manifest.selection!),
retries: retriesForFiles(files), timeoutMs: budget.timeoutMs, withinShardConcurrency: options.withinShardConcurrency,
tier: manifest.tier, profile, env });
},
} : {}),
env: {
...process.env,
@@ -1821,8 +2162,8 @@ async function main(): Promise<number> {
sliceIndex: options.sliceIndex,
sliceCount: manifest.sliceCount,
...(options.timeoutExplicit ? { timeoutOverrideMs: options.timeoutMs } : {}),
outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget }) =>
({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}) })),
outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused }) =>
({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}), ...(reused ? { reused } : {}) })),
};
fs.mkdirSync(evalDirBase, { recursive: true });
const sliceResultPath = path.join(evalDirBase, `slice-${options.sliceIndex}.json`);
@@ -1832,8 +2173,16 @@ async function main(): Promise<number> {
return summaryExitCode(summary);
}
if (options.listOnly && options.sliceBudgetMs !== null) {
const manifest = buildRunManifest({ tier: options.tier, profile: options.profile, sliceBudgetMs: options.sliceBudgetMs,
jobs: options.jobs, evalsAll: process.env.EVALS_ALL === '1', timeoutMs: timeoutOverride });
console.log(`[test:paid] slice plan preview: tier=${manifest.tier} profile=${manifest.profile ?? 'full'} (${manifest.selectionReason})`);
for (const line of formatSlicePlan(manifest)) console.log(line);
return 0;
}
const { selected, excluded } = selectPaidTestFiles(discovered, options.tier);
const shards = planPaidShards(selected, { maxFilesPerShard: options.maxFilesPerShard });
const shards = planPaidShards(expandCaseShards(selected, options.tier), { maxFilesPerShard: options.maxFilesPerShard });
// Parent-side diff selection (D9): skip whole shards whose mapped tests are
// all unselected. Fail-open everywhere — the child's self-skip stays
@@ -1862,7 +2211,7 @@ async function main(): Promise<number> {
const key = shards[index].join(' ');
const note = skipReasons.has(key) ? ` [would skip: ${skipReasons.get(key)}]` : '';
const budget = resolvePaidShardBudget(shards[index], options.timeoutExplicit ? options.timeoutMs : undefined);
console.log(` shard ${index + 1}/${shards.length}: ${key} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'}${note}`);
console.log(` shard ${index + 1}/${shards.length}: ${key} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'} retries=${retriesForFiles(shards[index])}${note}`);
}
if (excluded.length > 0) {
console.log(`\nExcluded (${excluded.length}):`);
+6 -4
View File
@@ -5,7 +5,7 @@ import * as path from 'node:path';
import { runPaidShard, shardSlug } from '../scripts/test-paid-shards';
const ROOT = path.resolve(import.meta.dir, '..');
const workflows = ['evals.yml', 'evals-periodic.yml'].map(name => ({
const workflows = ['evals.yml', 'evals-periodic.yml', 'evals-marathon.yml'].map(name => ({
name,
value: Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows', name), 'utf8')) as any,
}));
@@ -23,7 +23,7 @@ function render(template: string, fields: Record<string, string>): string {
test('every direct CI paid executor binds a safe unique run/attempt/job/slice identity', () => {
expect(executors.map(({ name, jobName }) => `${name}:${jobName}`)).toEqual([
'evals.yml:eval-slices', 'evals-periodic.yml:eval-slices', 'evals-periodic.yml:gate-census',
'evals.yml:eval-slices', 'evals-periodic.yml:eval-slices', 'evals-periodic.yml:gate-census', 'evals-marathon.yml:eval-slices',
]);
const ids = new Set<string>();
for (const [workflowIndex, { job, step }] of executors.entries()) {
@@ -32,9 +32,11 @@ test('every direct CI paid executor binds a safe unique run/attempt/job/slice id
expect(env.EVALS_RUN_ID).toBeString();
for (const run of ['36302678692', '36302678693']) {
for (const attempt of ['1', '2']) {
for (const slice of job.strategy.matrix.slice) {
// The planner sizes the matrix; cover more slices than any live plan.
expect(job.strategy.matrix.slice).toMatch(/^\$\{\{ fromJSON\(needs\.plan-slices\.outputs\.(?:[a-z]+_)?slices\) \}\}$/);
for (let slice = 1; slice <= 64; slice++) {
const id = render(env.EVALS_RUN_ID, {
'github.run_id': `${run}${workflowIndex === 0 ? '0' : '1'}`,
'github.run_id': `${run}${workflowIndex}`,
'github.run_attempt': attempt, 'matrix.slice': String(slice),
});
expect(id).toMatch(/^[A-Za-z0-9_-]+$/);
+7 -2
View File
@@ -30,8 +30,10 @@ test('the actual CI cookie repair planner executes only eight dependent cases wi
expect(manifest.evalsAll).toBe(false);
expect(manifest.selection).toEqual({ e2e: ['browse-basic', 'browse-snapshot', 'qa-quick', 'qa-only-no-fix', 'design-review-detector-shim-dom', 'diagram-triplet', 'canary-workflow', 'benchmark-workflow'], judges: [] });
expect(manifest.entries.filter(entry => entry.status === 'planned').map(entry => entry.file).sort()).toEqual([
'test/skill-e2e-bws.test.ts', 'test/skill-e2e-deploy.test.ts', 'test/skill-e2e-design.test.ts', 'test/skill-e2e-diagram.test.ts', 'test/skill-e2e-qa-workflow.test.ts',
'test/skill-e2e-bws.test.ts', 'test/skill-e2e-deploy.test.ts', 'test/skill-e2e-design.test.ts#design-review-detector-shim-dom', 'test/skill-e2e-diagram.test.ts', 'test/skill-e2e-qa-workflow.test.ts',
]);
// The case-sharded design file runs only its one selected cookie case.
expect(manifest.entries.filter(entry => entry.file.startsWith('test/skill-e2e-design.test.ts#') && entry.status === 'skipped-by-diff').length).toBeGreaterThan(0);
});
test('the existing quality and behavior phases retain their complete separate shard census', () => {
@@ -42,7 +44,10 @@ test('the existing quality and behavior phases retain their complete separate sh
expect(quality.evalsAll).toBe(true);
expect(behavior.evalsAll).toBe(true);
expect(qualityFiles).toHaveLength(1);
expect(behaviorFiles).toHaveLength(45);
// 44 files (first-task-scaffold registers no gate case, so the gate lane
// skips it); the five case-sharded files contribute one shard per gate case.
expect(new Set(behaviorFiles.map(file => file.split('#')[0])).size).toBe(44);
expect(behaviorFiles).toHaveLength(62);
expect(behaviorFiles).toEqual(expect.arrayContaining([
'test/skill-e2e-qa-callers.test.ts',
'test/skill-e2e-qa-functional-fix.test.ts',
+145
View File
@@ -0,0 +1,145 @@
import { describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import {
e2eReuseEnvironment, e2eReuseLaneProblem, e2eShardIdentity, e2eShardInputFiles, prepareE2EShardReuse,
type E2EShardReuseRequest,
} from '../scripts/e2e-shard-reuse';
import { buildRunManifest, fileCaseRegistration, runPaidShard, verifySliceResults, type SliceResult } from '../scripts/test-paid-shards';
const ROOT = path.resolve(import.meta.dir, '..');
const FILE = 'test/skill-e2e-deploy.test.ts';
const scratch = fs.mkdtempSync(path.join(os.tmpdir(), 'e2e-reuse-'));
const bin = path.join(scratch, 'bin');
fs.mkdirSync(bin);
fs.writeFileSync(path.join(bin, 'claude'), '#!/bin/sh\necho "9.9.9 (Claude Code)"\n', { mode: 0o755 });
const laneEnv = (over: NodeJS.ProcessEnv = {}): NodeJS.ProcessEnv => ({
PATH: `${bin}${path.delimiter}${process.env.PATH}`, HOME: scratch,
EVALS_TIER: 'gate', EVALS_PROFILE: 'pr', EVALS: '1',
EVALS_CACHE_DIR: path.join(scratch, 'cache'), EVALS_CACHE_REPOSITORY: 'garrytan/gstack', EVALS_CACHE_PR: '42',
EVALS_CACHE_RUNTIME_ID: 'a'.repeat(64), GITHUB_RUN_ID: '1001', GITHUB_RUN_ATTEMPT: '1', ANTHROPIC_API_KEY: 'sk-fixture',
GSTACK_CLAUDE_CLI_VERSION: '9.9.9 (Claude Code)',
...over,
});
function request(over: Partial<E2EShardReuseRequest> = {}): E2EShardReuseRequest {
const { registered, known } = fileCaseRegistration(FILE, fs.readFileSync(path.join(ROOT, FILE), 'utf8'));
return { root: ROOT, key: FILE, file: FILE, caseIds: ['setup-deploy-workflow'], registeredIds: registered, registrationKnown: known,
casePattern: '(?:^|\\s)(?:setup-deploy-workflow)$', expectedCases: 1, retries: 0, timeoutMs: 1_800_000,
withinShardConcurrency: 2, tier: 'gate', profile: 'pr', env: laneEnv(), ...over };
}
describe('E2E shard reuse eligibility', () => {
test('only the same-PR fast profile with an immutable runtime and default endpoint may reuse', () => {
expect(e2eReuseLaneProblem(laneEnv(), 'pr')).toBeNull();
for (const [env, mode, problem] of [
[laneEnv(), 'full-fallback', 'Only the fast PR profile'],
[laneEnv(), undefined, 'Only the fast PR profile'],
[laneEnv({ EVALS_CACHE_PR: '' }), 'pr', 'same-PR cache scope'],
[laneEnv({ EVALS_CACHE_RUNTIME_ID: 'latest' }), 'pr', 'immutable runtime'],
[laneEnv({ EVALS_FRESH: '1' }), 'pr', 'Fresh validation'],
[laneEnv({ EVALS_TIER: 'periodic' }), 'pr', 'Fresh validation'],
[laneEnv({ EVALS_CACHE_PURPOSE: 'periodic' }), 'pr', 'execute fresh'],
[laneEnv({ EVALS_CACHE_PURPOSE: 'marathon' }), 'pr', 'execute fresh'],
[laneEnv({ EVALS_CACHE_PURPOSE: 'release' }), 'pr', 'execute fresh'],
[laneEnv({ NODE_OPTIONS: '--require x' }), 'pr', 'Preload'],
[laneEnv({ ANTHROPIC_BASE_URL: 'https://proxy.example' }), 'pr', 'Custom model endpoint'],
] as const) expect(e2eReuseLaneProblem(env, mode)).toContain(problem);
});
test('the identity binds the child environment except run-scoped transport, and never secret values', () => {
const env = e2eReuseEnvironment(laneEnv({ EVALS_RUN_ID: 'run-1', GSTACK_EVAL_DIR: '/tmp/x', EVALS_SELECTION_JSON: '{}', EVALS_MODEL: 'm', UNRELATED: 'x' }));
expect(env.ANTHROPIC_API_KEY).toBe('set');
expect(env.EVALS_MODEL).toBe('m');
for (const name of ['EVALS_RUN_ID', 'GSTACK_EVAL_DIR', 'EVALS_SELECTION_JSON', 'EVALS_CACHE_DIR', 'EVALS_CACHE_PR', 'UNRELATED', 'GITHUB_RUN_ID']) {
expect(env[name], name).toBeUndefined();
}
expect(JSON.stringify(env)).not.toContain('sk-fixture');
});
test('consumed files cover the test closure, every registered touchfile, the globals and the harness', () => {
const files = e2eShardInputFiles(request());
for (const file of [FILE, 'test/helpers/e2e-helpers.ts', 'scripts/test-paid-shards.ts', 'scripts/e2e-shard-reuse.ts',
'bun.lock', '.github/workflows/evals.yml', '.github/actions/register-gstack-skills/action.yml', '.github/docker/Dockerfile.ci',
'setup-deploy/SKILL.md.tmpl', 'test/helpers/touchfiles-data.ts']) expect(files, file).toContain(file);
expect(files).not.toContain('package.json');
expect(files.some(file => file.startsWith('node_modules/'))).toBe(true);
expect(() => e2eShardInputFiles({ ...request(), registeredIds: ['no-such-case'] })).not.toThrow();
});
test('unknown or unprovable inputs fail closed', () => {
expect(e2eShardIdentity(request()).status).toBe('eligible');
for (const [over, reason] of [
[{ retries: 1 }, 'first attempt'],
[{ registrationKnown: false }, 'statically complete'],
[{ caseIds: [] }, 'exactly known'],
[{ expectedCases: 2 }, 'exactly known'],
[{ caseIds: ['not-registered'] }, 'exactly known'],
[{ file: 'test/skill-llm-eval.test.ts', key: 'test/skill-llm-eval.test.ts' }, 'audited E2E file'],
[{ env: laneEnv({ PATH: path.join(scratch, 'empty') }) }, 'Claude CLI version is unknown'],
] as const) {
const result = e2eShardIdentity(request(over as Partial<E2EShardReuseRequest>));
expect(result.status, reason).toBe('ineligible');
expect(result.status === 'ineligible' ? result.reason : '').toContain(reason);
}
});
test('any consumed parameter, pin or runtime change is a different identity', () => {
const key = (over: Partial<E2EShardReuseRequest>) => {
const result = e2eShardIdentity(request(over));
if (result.status !== 'eligible') throw new Error(result.reason);
return result.identity.key;
};
const base = key({});
expect(key({})).toBe(base);
expect(key({ env: laneEnv({ EVALS_RUN_ID: 'another-run', GITHUB_RUN_ID: '9' }) })).toBe(base);
for (const over of [{ timeoutMs: 1_000 }, { withinShardConcurrency: 1 }, { casePattern: 'x' },
{ env: laneEnv({ EVALS_MODEL: 'other' }) }, { env: laneEnv({ EVALS_CACHE_RUNTIME_ID: 'b'.repeat(64) }) },
{ env: laneEnv({ EVALS_CACHE_PR: '43' }) }] as Array<Partial<E2EShardReuseRequest>>) expect(key(over)).not.toBe(base);
});
});
describe('E2E shard reuse through the runner', () => {
test('a fresh first-attempt pass publishes; identical inputs then reuse without launching; changed inputs run', async () => {
const env = laneEnv({ EVALS_CACHE_DIR: path.join(scratch, 'roundtrip') });
const first = prepareE2EShardReuse(request({ env }))!;
expect(first.lookup()).toBeNull();
first.publish();
const hit = prepareE2EShardReuse(request({ env }))!.lookup();
expect(hit?.source.runId).toBe('1001/1');
expect(prepareE2EShardReuse(request({ env: { ...env, EVALS_MODEL: 'changed' } }))!.lookup()).toBeNull();
expect(prepareE2EShardReuse(request({ env: { ...env, EVALS_FRESH: '1' } }))).toBeNull();
const evalDir = path.join(scratch, 'evals');
let launched = 0;
const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, evalDirBase: evalDir, env, log: () => {},
expectedCaseIds: { [FILE]: ['setup-deploy-workflow'] },
reuseFor: (files, childEnv) => prepareE2EShardReuse(request({ env: { ...childEnv } })),
commandFor: () => { launched++; return { command: process.execPath, args: ['-e', 'process.exit(1)'] }; } });
expect(launched).toBe(0);
expect(outcome).toMatchObject({ status: 'passed', exitCode: 0, executedTests: 1, skippedTests: 0, reused: { runId: '1001/1' } });
const recorded = JSON.parse(fs.readFileSync(path.join(evalDir, 'shards', 'skill-e2e-deploy', 'e2e-reused-skill-e2e-deploy.json'), 'utf8'));
expect(recorded.tests).toEqual([expect.objectContaining({ name: 'setup-deploy-workflow', passed: true, execution: 'reused' })]);
});
test('a failed shard never publishes a receipt', async () => {
let published = 0;
const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, env: laneEnv(), log: () => {},
reuseFor: () => ({ lookup: () => null, publish: () => { published++; } }),
commandFor: () => ({ command: process.execPath, args: ['-e', 'process.exit(1)'] }) });
expect(outcome.status).toBe('failed');
expect(published).toBe(0);
});
test('the report accepts reused results only in the fast PR profile', () => {
const manifest = buildRunManifest({ tier: 'gate', sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } });
const planned = manifest.entries.filter(entry => entry.status === 'planned');
const reused = { inputKey: 'c'.repeat(64), runId: '1001/1', revision: 'd'.repeat(40), completedAt: 1 };
const results: SliceResult[] = [{ version: 1, tier: 'gate', sliceIndex: 1, sliceCount: 1, outcomes: planned.map(entry => ({
files: [entry.file], status: 'passed' as const, exitCode: 0, elapsedMs: 0, executedTests: 1, skippedTests: 0,
...(entry.budget ? { budget: entry.budget } : {}), ...(entry.file === FILE ? { reused } : {}) })) }];
expect(verifySliceResults(manifest, results).problems).toContain(`${FILE}: only the fast PR profile may reuse results; this lane executes fresh`);
});
});
+2 -2
View File
@@ -184,8 +184,8 @@ describe('E2E tier alignment (touchfiles declaration vs test self-gate)', () =>
// Both self-gate shapes count: the raw predicate and the consolidated
// helper (test/helpers/e2e-gate.ts documents this file as a consumer
// that must recognize describeE2ETier/e2eTierEnabled).
const selfGated = /EVALS_TIER\s*===\s*['"](gate|periodic)['"]/.test(content)
|| /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic)['"]/.test(content);
const selfGated = /EVALS_TIER\s*===\s*['"](gate|periodic|marathon)['"]/.test(content)
|| /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic|marathon)['"]/.test(content);
if (!usesNameSelection && selfGated) continue; // fail-open-safe standalone
invisible.push(
+38 -32
View File
@@ -1,5 +1,5 @@
import { expect, test } from 'bun:test';
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, OVERLAY_MAX_ACTIVE_SHARDS, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS } from '../scripts/test-paid-shards';
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards';
import { FINDING_RETRY_BUDGETS, ALL_TIERS, SHARD_RESERVE_MS } from './helpers/eval-budgets';
import fs from 'node:fs';
import os from 'node:os';
@@ -8,7 +8,8 @@ import path from 'node:path';
for (const budget of FINDING_RETRY_BUDGETS) {
test(`${budget.file}: supervision preserves every existing attempt and retry`, () => {
expect(budget.testMs).toBe(1_500_000);
expect(budget.retries).toBe(1);
// A 25-minute case is past RETRY_MAX_CASE_MS: a timed-out attempt is its verdict.
expect(budget.retries).toBe(0);
expect(retriesForFiles([budget.file])).toBe(budget.retries);
expect(budget.shardReserveMs).toBe(SHARD_RESERVE_MS);
expect(budget.shardMs).toBe(budget.cases * budget.testMs * (budget.retries + 1) + budget.shardReserveMs);
@@ -24,9 +25,8 @@ for (const budget of FINDING_RETRY_BUDGETS) {
expect([...source.matchAll(/timeoutMs:\s*1_500_000\b/g)]).toHaveLength(budget.cases);
}
expect([...source.matchAll(/1_500_000\s*\/\* physical ceiling:/g)]).toHaveLength(budget.cases);
// Current periodic CI already supports this supervision wall.
const workflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any;
expect(budget.shardMs).toBeLessThan(workflow.jobs['eval-slices']['timeout-minutes'] * 60_000);
// The planned periodic CI job cap supports this supervision wall.
expect(budget.shardMs).toBeLessThan(livePlan().plan!.ciTimeoutMinutes * 60_000);
});
test(`${budget.file}: own-shard allocation leaves ordinary and explicit limits intact`, () => {
@@ -104,31 +104,36 @@ test('actual shard launcher honors the explicit saved planner limit without a pr
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
}, 10000);
const periodicWorkflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any;
const periodicJob = periodicWorkflow.jobs['eval-slices'];
const periodicPlanStep = periodicWorkflow.jobs['plan-slices'].steps.find((step: any) => step.run?.includes('--tier periodic --emit-plan'));
const periodicSliceCount = Number(periodicPlanStep.run.match(/--slices\s+(\d+)/)?.[1]);
const periodicRunStep = periodicJob.steps.find((step: any) => step.run?.includes('--plan /tmp/paid-plan/manifest.json'));
const periodicWorkers = Number(periodicRunStep.env.EVALS_JOBS);
const livePlan = (discovered?: string[]) => buildRunManifest({ tier: 'periodic', sliceCount: periodicSliceCount,
evalsAll: true, env: { EVALS_ALL: '1' }, discovered });
function periodicLane() {
const workflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any;
const job = workflow.jobs['eval-slices'];
const planStep = workflow.jobs['plan-slices'].steps.find((step: any) => step.run?.includes('--tier periodic --emit-plan'));
const planned = parseCliOptions(planStep.run.slice(planStep.run.indexOf('scripts/test-paid-shards.ts') + 'scripts/test-paid-shards.ts'.length).trim().split(/\s+/), {});
const runStep = job.steps.find((step: any) => step.run?.includes('--plan /tmp/paid-plan/manifest.json'));
return { job, planStep, planned, workers: Number(runStep.env.EVALS_JOBS) };
}
function livePlan(discovered?: string[]) {
const { planned } = periodicLane();
return buildRunManifest({ tier: 'periodic', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs,
evalsAll: true, env: { EVALS_ALL: '1' }, discovered });
}
test('live periodic census fits the declared CI wall including setup', () => {
const { job, planStep, planned, workers } = periodicLane();
const m = livePlan();
expect(periodicPlanStep.run).not.toContain('--autoplan-slice');
expect(periodicJob.strategy.matrix.slice).toEqual(Array.from({ length: periodicSliceCount }, (_, index) => index + 1));
expect(periodicWorkers).toBe(2);
const walls = Array.from({ length: periodicSliceCount }, (_, index) => {
const files = m.entries.filter(e => e.status === 'planned' && e.slice === index + 1).map(e => e.file);
const workers = files.some(isOverlayTestFile) ? Math.min(periodicWorkers, OVERLAY_MAX_ACTIVE_SHARDS) : periodicWorkers;
return paidShardWallUpperBoundMs(files, workers);
});
expect(Math.max(...walls)).toBe(14_680_000);
expect(periodicJob['timeout-minutes']).toBe(360);
expect(periodicJob.strategy['max-parallel']).toBe(8);
expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(periodicJob['timeout-minutes'] * 60_000);
expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(70);
const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount);
expect(planStep.run).not.toContain('--autoplan-slice');
expect(job.strategy.matrix.slice).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}');
expect(job['timeout-minutes']).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}');
expect(workers).toBe(2);
expect(planned.jobs).toBe(workers);
const walls = Array.from({ length: m.sliceCount }, (_, index) => sliceSupervisedWallMs(sliceExecutionOrder(
m.entries.filter(e => e.status === 'planned' && e.slice === index + 1)).map(e => e.file), workers));
expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(m.plan!.ciTimeoutMinutes * 60_000);
expect(m.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
expect(m.sliceCount).toBeLessThanOrEqual(job.strategy['max-parallel']);
const plannedFiles = new Set(m.entries.filter(e => e.status === 'planned').map(e => shardFile(e.file)));
expect(plannedFiles).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected));
const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === m.sliceCount);
expect(overlays).toHaveLength(4);
expect(overlays.every(e => isOverlayTestFile(e.file))).toBe(true);
});
@@ -139,8 +144,8 @@ test('registered allocation is deterministic and preserves every discovered file
expect(files).toContain('test/skill-e2e-ship-skip.test.ts');
const m = livePlan(files);
expect(livePlan([...files].reverse())).toEqual(m);
expect(m.entries.map(e => e.file).sort()).toEqual([...files].sort());
expect(new Set(m.entries.map(e => e.file)).size).toBe(files.length);
expect([...new Set(m.entries.map(e => shardFile(e.file)))].sort()).toEqual([...files].sort());
expect(new Set(m.entries.map(e => e.file)).size).toBe(m.entries.length);
});
test('ordinary-only manifests retain round-robin allocation', () => {
@@ -171,17 +176,18 @@ test('single-slice manifest retains all registered files with one allocation', (
test('current detach supervision covers the live-census floor', () => {
const floorFor = (tier: 'gate' | 'periodic') => {
const files = selectPaidTestFiles(collectPaidTestFiles(), tier).selected;
// Case-sharded files contribute one shard per case, exactly as the runner plans.
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier);
const excess = files.reduce((n, file) => n + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - DEFAULT_SHARD_TIMEOUT_MS), 0);
return Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05);
};
const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8'));
const periodicTimeout = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]);
const gateTimeout = Number(pkg.scripts['eval:bg:gate'].match(/--timeout\s+(\d+)/)[1]);
expect(floorFor('gate')).toBe(42_851);
expect(floorFor('gate')).toBe(26_597);
expect(gateTimeout).toBe(49_320);
expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate'));
expect(floorFor('periodic')).toBe(37_727);
expect(floorFor('periodic')).toBe(30_797);
});
for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => {
+8 -5
View File
@@ -24,9 +24,10 @@ import {
DEFAULT_JOBS,
DEFAULT_SHARD_TIMEOUT_MS,
resolvePaidShardBudget,
expandCaseShards,
type PaidTier,
} from '../scripts/test-paid-shards';
import { FINDING_RETRY_BUDGETS } from './helpers/eval-budgets';
import { FILE_RETRY_BUDGETS } from './helpers/eval-budgets';
const ROOT = path.resolve(import.meta.dir, '..');
// 5% margin over the theoretical bound: detach setup, lock wait, aggregation.
@@ -57,7 +58,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
['periodic', 'eval:bg:periodic'],
] as Array<[PaidTier, string]>) {
test(`${script} covers ordinary ${tier} waves plus registered excess x ${MARGIN}`, () => {
const files = selectPaidTestFiles(collectPaidTestFiles(), tier).selected;
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier);
expect(files.length).toBeGreaterThan(0);
const floor = Math.ceil(worstCaseSeconds(files) * MARGIN);
const configured = detachTimeoutSeconds(script);
@@ -78,7 +79,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
const pkg = JSON.parse(fs.readFileSync(path.join(ROOT, 'package.json'), 'utf8'));
const jobs = Number(pkg.scripts['test:pr'].match(/EVALS_JOBS=\$\{EVALS_JOBS:-(\d+)\}/)?.[1]);
expect(jobs).toBe(2);
const files = selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected;
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected, 'gate');
const floor = Math.ceil(worstCaseSeconds(files, jobs) * MARGIN);
expect(detachTimeoutSeconds('eval:bg:pr')).toBeGreaterThanOrEqual(floor);
});
@@ -86,7 +87,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
test('eval:bg:release covers both complete tiers and their existing margins', () => {
const files = collectPaidTestFiles();
const floor = (['gate', 'periodic'] as const).reduce((sum, tier) =>
sum + Math.ceil(worstCaseSeconds(selectPaidTestFiles(files, tier).selected) * MARGIN), 0);
sum + Math.ceil(worstCaseSeconds(expandCaseShards(selectPaidTestFiles(files, tier).selected, tier)) * MARGIN), 0);
expect(detachTimeoutSeconds('eval:bg:release')).toBeGreaterThanOrEqual(floor);
});
});
@@ -94,7 +95,9 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
// One long job and one ordinary job can run side by side; the long job still
// needs its whole wall, regardless of the number of ordinary workers.
test('a heterogeneous pair rejects the old uniform-wall floor', () => {
const pair = [FINDING_RETRY_BUDGETS[0]!.file, 'test/skill-e2e-other.test.ts'];
// The longest registered wall (single-attempt finding files now fit the ordinary wall).
const longest = [...FILE_RETRY_BUDGETS].sort((a, b) => b.shardMs - a.shardMs)[0]!;
const pair = [longest.file, 'test/skill-e2e-other.test.ts'];
const actualLongest = Math.max(...pair.map(file => resolvePaidShardBudget([file]).timeoutMs)) / 1000;
expect(worstCaseSeconds(pair, 2)).toBe(actualLongest);
expect(worstCaseSeconds(pair, 2)).toBeGreaterThan(DEFAULT_SHARD_TIMEOUT_MS / 1000);
+119 -81
View File
@@ -22,25 +22,53 @@
import { describe, test, expect } from 'bun:test';
import * as fs from 'fs';
import * as path from 'path';
import { buildRunManifest, parseCliOptions, isOverlayTestFile, OVERLAY_MAX_ACTIVE_SHARDS, paidShardWallUpperBoundMs } from '../scripts/test-paid-shards';
import { buildRunManifest, parseCliOptions, sliceExecutionOrder, sliceSupervisedWallMs, CI_SETUP_ALLOWANCE_MINUTES } from '../scripts/test-paid-shards';
const ROOT = path.join(import.meta.dir, '..');
const read = (rel: string) => fs.readFileSync(path.join(ROOT, rel), 'utf-8');
const evalsYml = read('.github/workflows/evals.yml');
const periodicYml = read('.github/workflows/evals-periodic.yml');
const marathonYml = read('.github/workflows/evals-marathon.yml');
const registerAction = read('.github/actions/register-gstack-skills/action.yml');
/** Slice count the planner emits (`--slices N`) in a workflow source. */
function plannedSlices(source: string): number[] {
return [...source.matchAll(/--emit-plan\s+\S+\s+--slices\s+(\d+)/g)].map((m) => Number(m[1]));
/** Every planner site: its manifest path and budget (`--slice-budget S --jobs J`). */
function plannerSites(source: string): Array<{ manifest: string; budgetSeconds: number; jobs: number }> {
return [...source.matchAll(/--emit-plan\s+(\S+)\s+--slice-budget\s+(\d+)\s+--jobs\s+(\d+)/g)]
.map((m) => ({ manifest: m[1]!, budgetSeconds: Number(m[2]), jobs: Number(m[3]) }));
}
/** The executor matrix's slice list (`slice: [1, 2, ...]`). */
function matrixSlices(source: string): number[][] {
return [...source.matchAll(/^\s+slice: \[([\d,\s]+)\]\s*$/gm)].map((m) =>
m[1].split(',').map((n) => Number(n.trim())),
);
type Step = { id?: string; name?: string; run?: string; env?: Record<string, string>; with?: Record<string, string> };
type Job = { needs?: string[]; env?: Record<string, string>; outputs?: Record<string, string>; 'timeout-minutes': string | number;
strategy?: { 'max-parallel': number; matrix: { slice: string } }; steps: Step[] };
/**
* An executor's matrix and timeout must come from the planner step that wrote
* the manifest it downloads: `slices` from `[range(1; .sliceCount + 1)]` and
* `timeout-minutes` from `.plan.ciTimeoutMinutes`, never hand-written numbers.
*/
function expectPlannedExecutor(source: string, executorName: string, prefix: string) {
const workflow = Bun.YAML.parse(source) as { jobs: Record<string, Job> };
const planner = workflow.jobs['plan-slices']!;
const executor = workflow.jobs[executorName]!;
expect(executor.needs).toContain('plan-slices');
expect(executor.strategy!.matrix.slice).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}slices) }}`);
expect(executor['timeout-minutes']).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}timeout_minutes) }}`);
const [stepId] = /^\$\{\{ steps\.([\w-]+)\.outputs\.slices \}\}$/.exec(planner.outputs![`${prefix}slices`]!)!.slice(1);
expect(planner.outputs![`${prefix}timeout_minutes`]).toBe(`\${{ steps.${stepId}.outputs.timeout_minutes }}`);
const matrixStep = planner.steps.find(step => step.id === stepId)!;
const manifest = /jq -c '\[range\(1; \.sliceCount \+ 1\)\]' (\S+)\)/.exec(matrixStep.run!)![1]!;
expect(matrixStep.run).toContain(`jq -e '.plan.ciTimeoutMinutes' ${manifest})`);
const emit = planner.steps.filter(step => step.run?.includes(`--emit-plan ${manifest} `));
expect(emit).toHaveLength(1);
const execute = executor.steps.filter(step => step.run?.includes('--plan '));
expect(execute).toHaveLength(1);
expect(execute[0]!.run).toContain(`--plan ${manifest} --slice \${{ matrix.slice }}`);
expect(executor.steps.some(step => step.with?.path === manifest.replace(/\/manifest\.json$/, ''))).toBe(true);
// The planner packs for exactly the executor's worker count.
const site = plannerSites(emit[0]!.run!)[0]!;
expect(execute[0]!.env?.EVALS_JOBS).toBe(String(site.jobs));
return { site, emit: emit[0]!, execute: execute[0]!, executor, planner };
}
describe('evals.yml sliced-lane wiring (post-matrix)', () => {
@@ -67,13 +95,12 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => {
expect(evalsYml).toMatch(/EVALS_TIER=gate bun --no-install run scripts\/test-paid-shards\.ts --tier gate --report /);
});
test('executor matrix slice list matches the planner --slices count', () => {
const planned = plannedSlices(evalsYml);
const matrices = matrixSlices(evalsYml);
expect(planned, 'expected exactly one --emit-plan site in evals.yml').toHaveLength(1);
expect(matrices, 'expected exactly one slice matrix in evals.yml').toHaveLength(1);
const n = planned[0];
expect(matrices[0]).toEqual(Array.from({ length: n }, (_, i) => i + 1));
test('executor matrix and timeout come from the one budget planner', () => {
expect(plannerSites(evalsYml), 'expected exactly one --emit-plan site in evals.yml').toHaveLength(1);
const { site } = expectPlannedExecutor(evalsYml, 'eval-slices', '');
expect(site).toEqual({ manifest: '/tmp/paid-plan/manifest.json', budgetSeconds: 540, jobs: 2 });
// The validation-phase planner writes the same manifest with the same budget.
expect(evalsYml).toContain('sliceBudgetMs: 540000, jobs: 2');
});
test('reconcile exit is captured via PIPESTATUS, never $? after a pipe', () => {
@@ -81,7 +108,7 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => {
// `$?` after `... | tee` is tee's exit — always 0. That made the
// fail-closed reconcile gate silently fail-open (ship review army,
// 2026-08-31). Both lanes must read PIPESTATUS[0].
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) {
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) {
const reconcileBlocks = [...source.matchAll(/--report[^\n]*\| tee[^\n]*\n([\s\S]{0,400}?)GITHUB_OUTPUT/g)];
expect(reconcileBlocks.length, `${name}: expected a tee'd reconcile step`).toBeGreaterThanOrEqual(1);
for (const block of reconcileBlocks) {
@@ -124,78 +151,89 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => {
});
describe('evals-periodic.yml sliced-lane wiring', () => {
test('the CI job cap covers the live periodic slice census plus setup', () => {
type Env = Record<string, string>;
const workflow = Bun.YAML.parse(periodicYml) as {
env?: Env;
jobs: Record<string, {
env?: Env;
'timeout-minutes': number;
strategy?: { matrix: { slice: number[] } };
steps: Array<{ run?: string; env?: Env }>;
}>;
};
const planner = workflow.jobs['plan-slices'];
const executor = workflow.jobs['eval-slices'];
const plannerSteps = planner.steps.filter(step => step.run?.includes('EVALS_TIER=periodic ') && step.run.includes('--emit-plan '));
const executorSteps = executor.steps.filter(step => step.run?.includes('--plan '));
expect(plannerSteps).toHaveLength(1);
expect(executorSteps).toHaveLength(1);
const cliArgs = (run: string) => {
const command = /\bbun(?: --no-install)? run scripts\/test-paid-shards\.ts /.exec(run);
expect(command).not.toBeNull();
return run.slice(command!.index + command![0].length)
.replace(/\$\{\{\s*matrix\.slice\s*\}\}/g, '1').trim().split(/\s+/);
};
const plannerEnv = { ...workflow.env, ...planner.env, ...plannerSteps[0].env };
const plannerOptions = parseCliOptions(cliArgs(plannerSteps[0].run!), plannerEnv);
const executorOptions = parseCliOptions(cliArgs(executorSteps[0].run!), {
...workflow.env, ...executor.env, ...executorSteps[0].env,
});
expect(plannerEnv.EVALS_ALL).toBe('1');
expect(plannerOptions.tier).toBe('periodic');
expect(executorOptions.tier).toBe('periodic');
const slices = executor.strategy!.matrix.slice;
expect(slices).toEqual(Array.from({ length: plannerOptions.slices }, (_, i) => i + 1));
const manifest = buildRunManifest({
tier: plannerOptions.tier, sliceCount: plannerOptions.slices,
evalsAll: true, env: plannerEnv, rootDir: ROOT,
});
// Resolve the same per-file walls and overlay admission limit as execution.
const explicitWall = executorOptions.timeoutExplicit ? executorOptions.timeoutMs : undefined;
const setupAllowanceMinutes = 20;
const allowances = slices.map(slice => {
const files = manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice).map(entry => entry.file);
const normal = files.filter(file => !isOverlayTestFile(file));
const overlay = files.filter(isOverlayTestFile);
const bound = (group: string[], jobs: number) => paidShardWallUpperBoundMs(group, jobs, explicitWall);
return (bound(normal, executorOptions.jobs) + bound(overlay, Math.min(executorOptions.jobs, OVERLAY_MAX_ACTIVE_SHARDS))) / 60_000;
});
expect(Math.max(...allowances)).toBeGreaterThan(0);
const requiredMinutes = Math.max(...allowances) + setupAllowanceMinutes;
expect(executor['timeout-minutes'],
`periodic slice allowances ${allowances.join(', ')} minutes + ${setupAllowanceMinutes} minutes setup require ${requiredMinutes} CI minutes`,
).toBeGreaterThanOrEqual(requiredMinutes);
});
const lanes = [
{ source: periodicYml, name: 'evals-periodic.yml', executor: 'eval-slices', prefix: 'periodic_', tier: 'periodic' },
{ source: periodicYml, name: 'evals-periodic.yml', executor: 'gate-census', prefix: 'gate_', tier: 'gate' },
{ source: evalsYml, name: 'evals.yml', executor: 'eval-slices', prefix: '', tier: 'gate' },
{ source: marathonYml, name: 'evals-marathon.yml', executor: 'eval-slices', prefix: '', tier: 'marathon' },
] as const;
test('planner/executor/report tier=periodic and slice counts agree', () => {
for (const lane of lanes) {
test(`${lane.name}:${lane.executor} — the planned CI job cap covers every slice's supervised wall plus setup, and every slice starts at once`, () => {
const { emit, execute, executor } = expectPlannedExecutor(lane.source, lane.executor, lane.prefix);
const cliArgs = (run: string) => {
const command = /\bbun(?: --no-install)? run scripts\/test-paid-shards\.ts /.exec(run);
expect(command).not.toBeNull();
return run.slice(command!.index + command![0].length)
.replace(/\$\{\{\s*matrix\.slice\s*\}\}/g, '1').trim().split(/\s+/);
};
const workflow = Bun.YAML.parse(lane.source) as { env?: Record<string, string> };
// The complete census (EVALS_ALL) is the largest plan any event can produce.
const plannerEnv = { ...workflow.env, ...emit.env, EVALS_ALL: '1', EVALS_PROFILE: 'full' };
const planned = parseCliOptions(cliArgs(emit.run!), plannerEnv);
const active = parseCliOptions(cliArgs(execute.run!), { ...workflow.env, ...executor.env, ...execute.env, EVALS_PROFILE: 'full' });
expect(planned.tier).toBe(lane.tier);
expect(active.tier).toBe(lane.tier);
expect(active.jobs).toBe(planned.jobs);
const manifest = buildRunManifest({ tier: planned.tier, profile: 'full', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs,
evalsAll: true, env: plannerEnv, rootDir: ROOT, skipJudges: planned.skipJudges });
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === i + 1)).map(entry => entry.file), planned.jobs));
const requiredMinutes = Math.ceil(Math.max(0, ...walls) / 60_000) + CI_SETUP_ALLOWANCE_MINUTES;
expect(CI_SETUP_ALLOWANCE_MINUTES).toBe(20);
expect(manifest.plan!.ciTimeoutMinutes, `slice walls ${walls.join(', ')}ms`).toBe(requiredMinutes);
// GitHub-hosted-style job ceiling: a plan past it must be split, not truncated.
expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
expect(manifest.sliceCount, `${lane.name}:${lane.executor} plans more slices than max-parallel starts at once`)
.toBeLessThanOrEqual(executor.strategy!['max-parallel']);
});
}
test('planner/executor/report tier=periodic agree and plan with the ~9-minute budget', () => {
expect(periodicYml).toMatch(/EVALS_TIER=periodic bun --no-install run scripts\/test-paid-shards\.ts --tier periodic --emit-plan/);
expect(periodicYml).toMatch(/EVALS_TIER=periodic bun run scripts\/test-paid-shards\.ts --tier periodic --plan .* --slice /);
expect(periodicYml).toMatch(/EVALS_TIER=periodic bun --no-install run scripts\/test-paid-shards\.ts --tier periodic --report /);
const planned = plannedSlices(periodicYml);
const matrices = matrixSlices(periodicYml);
// Periodic work and the full gate census have distinct immutable plans.
expect(planned).toHaveLength(2);
expect(matrices).toHaveLength(2);
for (const [index, count] of planned.entries()) {
expect(matrices[index]).toEqual(Array.from({ length: count }, (_, i) => i + 1));
expect(plannerSites(periodicYml)).toEqual([
{ manifest: '/tmp/paid-plan/manifest.json', budgetSeconds: 540, jobs: 2 },
{ manifest: '/tmp/gate-census-plan/manifest.json', budgetSeconds: 540, jobs: 2 },
]);
});
});
describe('evals-marathon.yml non-blocking lane', () => {
const workflow = Bun.YAML.parse(marathonYml) as { on: Record<string, unknown>; env: Record<string, string>; jobs: Record<string, Job> };
test('runs weekly and on dispatch, always fresh, with its own fail-closed report and tracking issue', () => {
expect(Object.keys(workflow.on).sort()).toEqual(['schedule', 'workflow_dispatch']);
expect(workflow.env).toMatchObject({ EVALS_PROFILE: 'full', EVALS_FRESH: '1', EVALS_CACHE_PURPOSE: 'marathon' });
expect(marathonYml).not.toContain('actions/cache');
expect(marathonYml).toMatch(/EVALS_TIER=marathon bun --no-install run scripts\/test-paid-shards\.ts --tier marathon --emit-plan \/tmp\/marathon-plan\/manifest\.json --slice-budget 1 --jobs 1/);
expect(marathonYml).toMatch(/EVALS_TIER=marathon bun run scripts\/test-paid-shards\.ts --tier marathon --plan .* --slice /);
const report = workflow.jobs.report!;
expect(report.needs).toEqual(['plan-slices', 'eval-slices']);
const reconcile = report.steps.find(step => step.id === 'reconcile')!;
expect(reconcile.run).toContain('EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --report /tmp/marathon-report');
const guards = report.steps.filter(step => /Upsert tracking|Fail the workflow/.test(step.name ?? ''));
expect(guards).toHaveLength(2);
for (const step of guards) {
expect((step as { if?: string }).if).toContain("steps.reconcile.outputs.exit != '0'");
expect((step as { if?: string }).if).toContain("needs.eval-slices.result != 'success'");
}
expect(marathonYml).toContain('Weekly marathon evals: red lane needs triage');
});
test('the blocking lanes never plan or execute the marathon tier', () => {
for (const source of [evalsYml, periodicYml]) {
expect(source).not.toContain('--tier marathon');
expect(source).not.toContain('EVALS_TIER=marathon');
}
});
});
describe('shared setup composites (both surviving lanes)', () => {
test('both lanes register skills through the shared composite', () => {
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) {
describe('shared setup composites (every paid lane)', () => {
test('every lane registers skills through the shared composite', () => {
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) {
expect(source, `${name} must use the register-gstack-skills composite`)
.toContain('uses: ./.github/actions/register-gstack-skills');
// No inline re-implementation creeping back beside the composite.
@@ -218,7 +256,7 @@ describe('shared setup composites (both surviving lanes)', () => {
for (const action of ['seed-claude-config', 'restore-deps', 'fix-bun-temp']) {
expect(fs.existsSync(path.join(ROOT, '.github', 'actions', action, 'action.yml')), `missing composite: ${action}`).toBe(true);
}
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) {
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) {
expect(source, `${name} must use seed-claude-config`).toContain('uses: ./.github/actions/seed-claude-config');
expect(source, `${name} must use restore-deps`).toContain('uses: ./.github/actions/restore-deps');
expect(source, `${name} must use fix-bun-temp`).toContain('uses: ./.github/actions/fix-bun-temp');
+103 -40
View File
@@ -46,70 +46,133 @@ export const ALL_TIERS = {
/** Supervision reserve added to every registered whole-file wall. */
export const SHARD_RESERVE_MS = 2 * 60_000;
/** Whole-file supervision must cover each existing attempt and its retry.
* These fixtures already allow 25 minutes per case; the old 30-minute
* wall could kill a second attempt after five minutes. No case budget grows.
/**
* Retry policy (approved 2026-09-29): a timed-out attempt is a verdict. Bun's
* --retry reruns a failed case after it may have spent its whole budget, so an
* automatic retry is kept only where one more attempt is short: every case of
* the file has a per-attempt budget of at most RETRY_MAX_CASE_MS, the CAPTURE
* tier plus its recording grace. Those failures are fast flake classes (API
* blips, tool hiccups) and a retry costs at most one more short attempt. Files
* with any longer case run once. Per-case budgets never change with this rule.
*/
export const RETRY_MAX_CASE_MS = CAPTURE_MS + 15_000;
export function retriesWithinCaseCap(caseMs: number, configuredRetries: number): number {
return caseMs <= RETRY_MAX_CASE_MS ? configuredRetries : 0;
}
/**
* Unregistered paid files that keep one automatic retry: every case budget is
* JUDGE or CAPTURE tier (test/paid-retry-supervision.test.ts scans each source).
* Registered rows below derive retries from their declared caseMs; every other
* paid file runs once.
*/
export const SHORT_CASE_RETRY_FILES: readonly string[] = [
'test/codex-e2e-sol-scope.test.ts',
'test/llm-judge-recommendation.test.ts',
'test/skill-e2e-ask-user-question-format-compliance.test.ts',
'test/skill-e2e-benchmark-providers.test.ts',
'test/skill-e2e-bws.test.ts',
'test/skill-e2e-context-skills.test.ts',
'test/skill-e2e-coverage-audit.test.ts',
'test/skill-e2e-diagram.test.ts',
'test/skill-e2e-first-task-scaffold.test.ts',
'test/skill-e2e-gbrain-roundtrip-local.test.ts',
'test/skill-e2e-hermetic-canary.test.ts',
'test/skill-e2e-investigate-owned-completion.test.ts',
'test/skill-e2e-investigate-owned-termination.test.ts',
'test/skill-e2e-learnings.test.ts',
'test/skill-e2e-plan-tune.test.ts',
'test/skill-e2e-qa-functional-fix.test.ts',
'test/skill-e2e-qa-functional.test.ts',
'test/skill-e2e-review-army.test.ts',
'test/skill-e2e-review.test.ts',
'test/skill-e2e-session-intelligence.test.ts',
'test/skill-e2e-setup-gbrain-bad-token.test.ts',
'test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts',
'test/skill-e2e-setup-gbrain-remote.test.ts',
'test/skill-e2e-ship-hook-consent.test.ts',
'test/skill-e2e-ship-hook-refresh.test.ts',
'test/skill-e2e-ship-skip.test.ts',
'test/skill-e2e-sync-gbrain-readiness.test.ts',
'test/skill-e2e-third-party-actions.test.ts',
'test/skill-e2e-triage.test.ts',
'test/skill-routing-e2e.test.ts',
];
/** Whole-file supervision covers every attempt the retry policy allows.
* These fixtures allow 25 minutes per case, so they run once.
* Reserve the sequential upper bound even when Bun runs sibling cases together.
*/
export const FINDING_RETRY_BUDGETS = [
{ file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 },
].map(({ file, cases }) => ({
file, cases,
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
testMs: 1_500_000,
retries: 1,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: cases * 1_500_000 * 2 + SHARD_RESERVE_MS,
}));
].map(({ file, cases }) => {
const retries = retriesWithinCaseCap(1_500_000, 1);
return {
file, cases,
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
testMs: 1_500_000,
caseMs: 1_500_000,
retries,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: cases * 1_500_000 * (retries + 1) + SHARD_RESERVE_MS,
};
});
/** Three existing captures and one configured retry; only supervision grows. */
/** Three existing captures in one 16-minute case, so the file runs once. */
export const AUQ_CONSISTENCY_RETRY_BUDGET = {
file: 'test/skill-e2e-auq-consistency.test.ts',
id: 'auq-consistency-existing-retry-v1',
cases: 1,
testMs: 3 * CAPTURE_MS + 60_000,
retries: 1,
caseMs: 3 * CAPTURE_MS + 60_000,
retries: retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1),
shardReserveMs: SHARD_RESERVE_MS,
shardMs: (3 * CAPTURE_MS + 60_000) * 2 + SHARD_RESERVE_MS,
shardMs: (3 * CAPTURE_MS + 60_000) * (retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1) + 1) + SHARD_RESERVE_MS,
} as const;
/** These fixtures have a fixed case count in every supported tier. */
export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET];
/** Whole-file walls cover all existing cases and retries, even if Bun runs them
* sequentially. Mixed-tier files reserve their larger complete tier, never a
* currently selected subset. These rows add no case-count or model-work policy.
* The 10-second terms preserve the existing Codex/recording finalization grace.
/** Whole-file walls cover all existing cases and every allowed attempt, even if
* Bun runs them sequentially. Mixed-tier files reserve their larger complete
* tier, never a currently selected subset. caseMs is the longest single case
* budget, which decides the retry (RETRY_MAX_CASE_MS). These rows add no
* case-count or model-work policy. The 10-second terms preserve the existing
* Codex/recording finalization grace.
*/
export const FILE_RETRY_BUDGETS = [
...STRICT_RETRY_CASE_BUDGETS,
...[
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), retries: 1 },
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, retries: 1 },
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 5 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, retries: 1 },
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 5 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
// Seventeen workflow judges include their 10s recording grace; the other
// seven judges retain 120s. Supervise all 24 and the existing one retry.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, retries: 1 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, retries: 2 },
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 2 },
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
// Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s.
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), retries: 1 },
].map(({ file, attemptMs, retries }) => ({
file, attemptMs, retries,
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
})),
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS, configuredRetries: 1 },
].map(({ file, attemptMs, caseMs, configuredRetries }) => {
const retries = retriesWithinCaseCap(caseMs, configuredRetries);
return {
file, attemptMs, caseMs, retries,
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
};
}),
];
/** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */
+19 -9
View File
@@ -22,17 +22,18 @@ const fakeEnv = {
describe('overlay file policy', () => {
test('grouped planning isolates every overlay and preserves ordinary retries', () => {
const workflow = 'test/skill-e2e-workflow.test.ts';
const files = [...overlayFiles, normalFile, workflow];
// Two short-case files keep their one retry (timeout-is-a-verdict rule).
const workflow = 'test/skill-e2e-review.test.ts';
const files = [...overlayFiles, 'test/skill-e2e-triage.test.ts', workflow];
for (const maxFilesPerShard of [2, 3, 10]) {
const shards = planPaidShards(files, { maxFilesPerShard });
expect(shards.flat().sort()).toEqual([...files].sort());
for (const file of overlayFiles) expect(shards).toContainEqual([file]);
const workflowShard = shards.find(shard => shard.includes(workflow))!;
expect(workflowShard.some(isOverlayTestFile)).toBe(false);
expect(retriesForFiles(workflowShard)).toBe(2);
expect(retriesForFiles(workflowShard)).toBe(1);
const args = buildPaidShardArgs(workflowShard, resolvePaidShardTimeoutMs(workflowShard), 2, retriesForFiles(workflowShard));
expect(args[args.indexOf('--retry') + 1]).toBe('2');
expect(args[args.indexOf('--retry') + 1]).toBe('1');
expect(planPaidShards(files.map(file => file.replaceAll('/', '\\')), { maxFilesPerShard })).toEqual(shards);
}
});
@@ -65,8 +66,10 @@ describe('overlay file policy', () => {
for (const file of [normalFile, 'test/skill-e2e-overlay-harness.test.ts', 'test/model-overlays.test.ts']) {
expect(isOverlayTestFile(file)).toBe(false);
expect(resolvePaidShardTimeoutMs([file])).toBe(DEFAULT_SHARD_TIMEOUT_MS);
expect(retriesForFiles([file])).toBe(1);
// Not overlays; unlisted files run once because their case budget is unknown.
expect(retriesForFiles([file])).toBe(0);
}
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1);
expect(resolvePaidShardTimeoutMs([normalFile], 1234)).toBe(1234);
expect(resolvePaidShardTimeoutMs([overlayFiles[0]], 1_900_000)).toBe(1_900_000);
expect(() => resolvePaidShardTimeoutMs([overlayFiles[0]], 1_800_000)).toThrow('explicit wall');
@@ -157,8 +160,8 @@ describe('overlay manifest affinity and CI capacity', () => {
const workflow = Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows/evals-periodic.yml'), 'utf8')) as {
jobs: Record<string, {
steps: Array<{ run?: string; env?: NodeJS.ProcessEnv }>;
strategy: { matrix: { slice: number[] } };
'timeout-minutes': number;
strategy: { matrix: { slice: string } };
'timeout-minutes': string;
}>;
};
const job = workflow.jobs['eval-slices'];
@@ -166,13 +169,20 @@ describe('overlay manifest affinity and CI capacity', () => {
const jobs = parseCliOptions([], step.env).jobs;
expect(jobs).toBe(2);
expect(parseCliOptions([], step.env).withinShardConcurrency).toBe(2);
expect(job.strategy.matrix.slice).toEqual([1, 2, 3, 4, 5, 6, 7]);
expect(job.strategy.matrix.slice).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}');
expect(job['timeout-minutes']).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}');
const normalMinutes = Math.ceil(18 / jobs) * resolvePaidShardTimeoutMs([normalFiles[0]]) / 60_000;
const overlayMinutes = Math.ceil(overlayFiles.length / OVERLAY_MAX_ACTIVE_SHARDS)
* Math.max(...overlayFiles.map(file => resolvePaidShardTimeoutMs([file]))) / 60_000;
expect(normalMinutes).toBe(270);
expect(overlayMinutes).toBe(122);
expect(job['timeout-minutes']).toBeGreaterThanOrEqual(Math.max(normalMinutes, overlayMinutes) + 20);
// The CI budget plan (what the workflow runs) keeps the four overlays
// in one final one-at-a-time slice and its job cap covers them.
const budget = buildRunManifest({ ...opts, sliceCount: undefined, sliceBudgetMs: 540_000, jobs });
const lastSlice = budget.entries.filter(e => e.slice === budget.sliceCount).map(e => e.file).sort();
expect(lastSlice).toEqual([...overlayFiles].sort());
expect(budget.plan!.ciTimeoutMinutes).toBeGreaterThanOrEqual(overlayMinutes + 20);
expect(budget.plan!.ciTimeoutMinutes).toBe(Math.ceil(overlayMinutes) + 20);
// Gate selection keeps its original periodic exclusion and all six
// ordinary slices; reservation does not spend an empty slot in gate.
+2 -1
View File
@@ -150,7 +150,8 @@ describe('PR profile paid-runner integration', () => {
expect(() => parseRunManifest(JSON.stringify(injected))).toThrow('outside its PR case selection');
for (const action of ['remove', 'skip', 'duplicate'] as const) {
const missing = structuredClone(manifest);
const file = 'test/skill-e2e-plan.test.ts';
// plan.test is case-sharded: its PR case runs as `<file>#<case id>`.
const file = manifest.entries.find(entry => entry.status === 'planned' && entry.file.startsWith('test/skill-e2e-plan.test.ts#'))!.file;
if (action === 'remove') missing.entries = missing.entries.filter(entry => entry.file !== file);
if (action === 'skip') missing.entries.find(entry => entry.file === file)!.status = 'skipped-by-diff';
if (action === 'duplicate') missing.entries.push({ ...missing.entries.find(entry => entry.file === file)! });
+92 -57
View File
@@ -4,34 +4,69 @@ import { join } from 'node:path';
import {
buildPaidShardArgs, buildRunManifest, parseRunManifest, planPaidShards,
DEFAULT_JOBS, parseCliOptions, paidShardWallUpperBoundMs, resolvePaidShardBudget, retriesForFiles, verifySliceResults, collectPaidTestFiles, selectPaidTestFiles,
shardFile, sliceExecutionOrder, sliceSupervisedWallMs, CASE_SHARDED_FILES,
} from '../scripts/test-paid-shards';
import {
ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS,
ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, RETRY_MAX_CASE_MS, SHORT_CASE_RETRY_FILES,
FINDING_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS,
} from './helpers/eval-budgets';
import { E2E_TOUCHFILES } from './helpers/touchfiles';
const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8');
const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file));
// Walls cover every attempt the retry rule allows: files with a case budget
// past RETRY_MAX_CASE_MS run once (a timed-out attempt is a verdict).
const expectedWalls = {
'test/skill-e2e-qa-callers.test.ts': 3_270_000,
'test/skill-e2e-shared-libs-paths.test.ts': 3_720_000,
'test/skill-e2e-ship-docsync.test.ts': 10_920_000,
'test/skill-e2e-shared-libs-paths.test.ts': 1_920_000,
'test/skill-e2e-ship-docsync.test.ts': 5_520_000,
'test/skill-llm-eval.test.ts': 6_220_000,
'test/skill-e2e-auq-consistency.test.ts': 2_040_000,
'test/skill-e2e-auq-consistency.test.ts': 1_080_000,
'test/skill-e2e-auq-matrix.test.ts': 3_720_000,
'test/skill-e2e-plan-format.test.ts': 2_600_000,
'test/skill-e2e-auto-decide-preserved.test.ts': 1_920_000,
'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_920_000,
'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_920_000,
'test/skill-e2e-plan-design-finding-floor.test.ts': 1_920_000,
'test/skill-e2e-plan-devex-finding-floor.test.ts': 1_920_000,
'test/skill-e2e-plan-mode-no-op.test.ts': 9_120_000,
'test/skill-e2e-plan-ceo-mode-routing.test.ts': 2_520_000,
'test/skill-e2e-plan-eng-plan-mode.test.ts': 2_520_000,
'test/skill-e2e-auto-decide-preserved.test.ts': 1_020_000,
'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_020_000,
'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_020_000,
'test/skill-e2e-plan-design-finding-floor.test.ts': 1_020_000,
'test/skill-e2e-plan-devex-finding-floor.test.ts': 1_020_000,
'test/skill-e2e-plan-mode-no-op.test.ts': 3_120_000,
'test/skill-e2e-plan-ceo-mode-routing.test.ts': 1_320_000,
'test/skill-e2e-plan-eng-plan-mode.test.ts': 1_320_000,
'test/skill-e2e-plan-prosons.test.ts': 2_600_000,
'test/skill-e2e-plan.test.ts': 7_320_000,
'test/skill-e2e-plan.test.ts': 3_720_000,
};
test('retry rule: only files whose every case is CAPTURE tier or shorter retry; longer cases run once', () => {
expect(RETRY_MAX_CASE_MS).toBe(ALL_TIERS.CAPTURE_MS + 15_000);
for (const row of FILE_RETRY_BUDGETS) {
expect(row.retries, row.file).toBe(row.caseMs <= RETRY_MAX_CASE_MS ? (row.file.endsWith('plan-mode-no-op.test.ts') ? 2 : 1) : 0);
expect(retriesForFiles([row.file])).toBe(row.retries);
}
expect(FILE_RETRY_BUDGETS.filter(row => row.retries > 0).map(row => row.file).sort()).toEqual([
'test/skill-e2e-auq-matrix.test.ts', 'test/skill-e2e-plan-format.test.ts', 'test/skill-e2e-plan-prosons.test.ts',
'test/skill-e2e-qa-callers.test.ts', 'test/skill-llm-eval.test.ts',
]);
const paid = collectPaidTestFiles();
for (const file of SHORT_CASE_RETRY_FILES) {
expect(paid, `stale SHORT_CASE_RETRY_FILES entry: ${file}`).toContain(file);
expect(FILE_RETRY_BUDGETS.some(row => row.file === file)).toBe(false);
const source = read(file);
// Declared short budgets only: a JUDGE/CAPTURE tier or a literal at most the
// cap, no longer tier and no ms literal past the cap.
const literals = [...source.matchAll(/(?<![\w.])(\d{1,3}(?:_\d{3})+|\d{5,})(?![\w.])/g)]
.map(match => Number(match[1]!.replace(/_/g, '')));
expect(/\b(?:JUDGE_MS|CAPTURE_MS)\b/.test(source) || literals.some(ms => ms >= 60_000 && ms <= RETRY_MAX_CASE_MS), file).toBe(true);
expect(source, file).not.toMatch(/\b(?:CAPTURE_LONG_MS|PTY_MS|PTY_LONG_MS|OVERLAY_CASE_[A-Z_]+)\b/);
expect(literals.filter(ms => ms > RETRY_MAX_CASE_MS && ms < 10_000_000), file).toEqual([]);
expect(retriesForFiles([file])).toBe(1);
}
for (const file of paid.filter(file => !SHORT_CASE_RETRY_FILES.includes(file) && !FILE_RETRY_BUDGETS.some(row => row.file === file))) {
expect(retriesForFiles([file]), file).toBe(0);
}
expect(retriesForFiles([SHORT_CASE_RETRY_FILES[0]!, 'test/skill-e2e-plan.test.ts'])).toBe(0);
});
test('registration covers exactly the seventeen demonstrated full-file retry gaps', () => {
expect(Object.fromEntries(newBudgets.map(row => [row.file, row.shardMs]))).toEqual(expectedWalls);
expect(new Set(FILE_RETRY_BUDGETS.map(row => row.file)).size).toBe(19);
@@ -86,12 +121,17 @@ test('source allowances retain all captures, cases, and finalization grace', ()
]);
});
// Case-sharded files plan one registered case per shard (`<file>#<case id>`).
const plannedKey = (file: string) => CASE_SHARDED_FILES.includes(file)
? `${file}#${Object.keys(E2E_TOUCHFILES).find(id => E2E_TOUCHFILES[id]!.includes(file))}` : file;
for (const row of newBudgets) {
const key = plannedKey(row.file);
const manifest = () => ({ version: 1 as const, tier: 'periodic' as const, evalsAll: false,
sliceCount: 1, entries: [{ file: row.file, slice: 1, status: 'planned' as const, budget: resolvePaidShardBudget([row.file]) }] });
sliceCount: 1, entries: [{ file: key, slice: 1, status: 'planned' as const, budget: resolvePaidShardBudget([key]) }] });
const result = (count = 1) => [{ version: 1 as const, tier: 'periodic' as const, sliceIndex: 1, sliceCount: 1,
outcomes: [{ files: [row.file], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count,
skippedTests: 0, budget: resolvePaidShardBudget([row.file]) }] }];
outcomes: [{ files: [key], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count,
skippedTests: 0, budget: resolvePaidShardBudget([key]) }] }];
test(`${row.file}: full wall and existing retries propagate through planning`, () => {
expect(retriesForFiles([row.file])).toBe(row.retries);
@@ -132,8 +172,11 @@ test('fixed AUQ count remains strict while mixed-tier files keep ordinary case h
[AUQ_CONSISTENCY_RETRY_BUDGET.file, 0, false],
[AUQ_CONSISTENCY_RETRY_BUDGET.file, 1, true],
[AUQ_CONSISTENCY_RETRY_BUDGET.file, 2, false],
['test/skill-e2e-plan.test.ts', 5, true],
['test/skill-e2e-plan.test.ts', 7, true],
['test/skill-e2e-ship-docsync.test.ts', 5, true],
['test/skill-e2e-ship-docsync.test.ts', 7, true],
// A case shard of a case-sharded registered file executes exactly its case.
[plannedKey('test/skill-e2e-plan.test.ts'), 1, true],
[plannedKey('test/skill-e2e-plan.test.ts'), 2, false],
] as const) {
const budget = resolvePaidShardBudget([file]);
const manifest: any = { version: 1, tier: 'periodic', evalsAll: true, sliceCount: 1,
@@ -158,7 +201,7 @@ test('quality judge supervision includes the added judge without changing ordina
expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS');
expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000);
expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([
...Array(2).fill([1, 1500000, 1, 3120000]),
...Array(2).fill([1, 1500000, 0, 1620000]),
]);
for (const tier of ['gate', 'periodic'] as const) {
const m = buildRunManifest({ tier, sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } });
@@ -184,7 +227,7 @@ test('detached PR fallback and release commands cover their actual default worke
const prFloor = Math.ceil((Math.ceil(fullGateFiles.length / prWorkers) * 1_800_000 + fullGateFiles.reduce(
(total, file) => total + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - 1_800_000), 0,
)) / 1000 * 1.05);
expect(prFloor).toBe(74_981);
expect(prFloor).toBe(71_957);
expect(prWall).toBe(92_820_000);
expect(prWall).toBeGreaterThanOrEqual(paidShardWallUpperBoundMs(files, prWorkers) + 120_000);
@@ -203,8 +246,8 @@ test('detached PR fallback and release commands cover their actual default worke
)) / 1000 * 1.05));
}
const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000;
expect(releaseFloors).toEqual([42_851, 37_727]);
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(80_578);
expect(releaseFloors).toEqual([26_597, 30_797]);
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(57_394);
expect(detachedReleaseWall).toBe(116_700_000);
expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000);
});
@@ -217,10 +260,10 @@ const cliOptions = (step: { run: string; env?: Record<string, string> }) => {
return parseCliOptions(args, step.env ?? {});
};
test('both gate executors cover the complete census without increasing aggregate workers', () => {
test('both gate executors plan the complete census and supervise every planned slice', () => {
const periodic: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml'));
const main: any = Bun.YAML.parse(read('.github/workflows/evals.yml'));
for (const [workflow, jobName, workers, slices] of [[main, 'eval-slices', 2, 7], [periodic, 'gate-census', 1, 7]] as const) {
for (const [workflow, jobName, prefix, skipJudges] of [[main, 'eval-slices', '', false], [periodic, 'gate-census', 'gate_', true]] as const) {
const planner = workflow.jobs['plan-slices'];
const executor = workflow.jobs[jobName];
const emit = planner.steps.filter((step: any) => step.run?.includes('EVALS_TIER=gate ') && step.run.includes('--emit-plan '));
@@ -230,41 +273,35 @@ test('both gate executors cover the complete census without increasing aggregate
const planned = cliOptions(emit[0]), active = cliOptions(execute[0]);
expect(planned.tier).toBe('gate');
expect(active.tier).toBe('gate');
expect(active.jobs).toBe(workers);
expect(active.jobs).toBe(2);
expect(planned.jobs).toBe(active.jobs);
expect(planned.sliceBudgetMs).toBe(540_000);
expect(planned.skipJudges).toBe(skipJudges);
expect(execute[0].env.EVALS_CONCURRENCY).toBe('2');
expect(executor.strategy['fail-fast']).toBe(false);
expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: slices }, (_, i) => i + 1));
expect(planned.slices).toBe(slices);
const manifest = buildRunManifest({ tier: 'gate', sliceCount: planned.slices, evalsAll: true, env: { EVALS_ALL: '1' } });
expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(46);
const files = manifest.entries.filter(row => row.status === 'planned').map(row => row.file);
expect(new Set(files).size).toBe(46);
expect(executor.strategy.matrix.slice).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}slices) }}`);
expect(executor['timeout-minutes']).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}timeout_minutes) }}`);
const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs, evalsAll: true, env: { EVALS_ALL: '1' }, skipJudges });
const files = [...new Set(manifest.entries.filter(row => row.status === 'planned').map(row => shardFile(row.file)))];
expect(files).toContain('test/skill-e2e-ship-skip.test.ts');
expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected.sort());
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
manifest.entries.filter(row => row.status === 'planned' && row.slice === slice).map(row => row.file), workers,
));
expect(executor['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected
.filter(file => !skipJudges || !file.startsWith('test/skill-llm-eval')).sort());
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
manifest.entries.filter(row => row.status === 'planned' && row.slice === i + 1)).map(row => row.file), active.jobs));
expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
expect(manifest.sliceCount).toBeLessThanOrEqual(executor.strategy['max-parallel']);
if (jobName === 'gate-census') {
expect(Math.max(...walls)).toBe(16_440_000);
expect(executor['timeout-minutes']).toBe(352);
expect(emit[0].env.EVALS_ALL).toBe('1');
expect(executor.strategy['max-parallel']).toBe(4);
expect(executor.strategy['max-parallel'] * active.jobs).toBe(4);
expect(emit[0].run).toContain('--emit-plan /tmp/gate-census-plan/manifest.json');
expect(execute[0].run).toContain('--plan /tmp/gate-census-plan/manifest.json --slice ${{ matrix.slice }}');
expect(executor.steps.some((step: any) => step.with?.name === 'gate-census-plan')).toBe(true);
expect(executor.steps.filter((step: any) => step.run?.includes('--emit-plan '))).toHaveLength(0);
} else {
expect(Math.max(...walls)).toBe(12_720_000);
expect(executor['timeout-minutes']).toBe(265);
expect(executor.strategy['max-parallel']).toBe(6);
expect(executor.strategy['max-parallel'] * active.jobs).toBe(12);
}
}
});
test('the periodic executor supervises every actual case and retry within its CI wall', () => {
test('the periodic executor supervises every actual case and retry within its planned CI wall', () => {
const workflow: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml'));
const executor = workflow.jobs['eval-slices'];
const emit = workflow.jobs['plan-slices'].steps.filter((step: any) =>
@@ -274,21 +311,19 @@ test('the periodic executor supervises every actual case and retry within its CI
expect(execute).toHaveLength(1);
const planned = cliOptions(emit[0]), active = cliOptions(execute[0]);
expect(planned.tier).toBe('periodic');
expect(planned.slices).toBe(7);
expect(planned.sliceBudgetMs).toBe(540_000);
expect(active.jobs).toBe(2);
expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: planned.slices }, (_, i) => i + 1));
const manifest = buildRunManifest({ tier: 'periodic', sliceCount: planned.slices,
expect(planned.jobs).toBe(active.jobs);
const manifest = buildRunManifest({ tier: 'periodic', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs,
evalsAll: true, env: { EVALS_ALL: '1' } });
const census = manifest.entries.filter(row => row.status === 'planned');
expect(census).toHaveLength(70);
expect(new Set(census.map(row => shardFile(row.file)))).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected));
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_220_000);
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
census.filter(row => row.slice === slice).map(row => row.file), active.jobs,
));
expect(Math.max(...walls)).toBe(14_680_000);
expect(executor.strategy['max-parallel']).toBe(8);
expect(executor['timeout-minutes']).toBe(360);
expect(executor['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
census.filter(row => row.slice === i + 1)).map(row => row.file), active.jobs));
expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
expect(manifest.sliceCount).toBeLessThanOrEqual(executor.strategy['max-parallel']);
});
test('gate census requires all seven distinct slice results and its own reconciliation', () => {
+110 -21
View File
@@ -6,8 +6,9 @@
* - per-slice selector divergence → ONE planner manifest, executors consume
* - hollow lanes → a slice with no artifact is a FAILURE, not an absence
* - hollow shards → EVALS_ALL + exit 0 + zero executed tests ≠ pass
* - retry parity → the old matrix rows' earned `retries: 2` survive as a
* literals map, not folklore
* - retry policy → a timed-out attempt is a verdict; only short-case files retry
* - budget packing → recorded work packs into ~9-minute executors whose count
* and CI job timeout come from the plan
*/
import { describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
@@ -21,13 +22,18 @@ import {
buildRunManifest,
loadPaidTestDurations,
mergePaidTestDurations,
packBySliceBudget,
estimatedSliceMs,
sliceExecutionOrder,
sliceSupervisedWallMs,
writePaidTestDurations,
CI_SETUP_ALLOWANCE_MINUTES,
paidShardWallUpperBoundMs,
parseCliOptions,
parseRunManifest,
resolvePaidShardBudget,
SUPERVISED_WORKER_COUNTS,
retriesForFiles,
RETRY_OVERRIDES,
summarize,
summaryExitCode,
verifySliceResults,
@@ -46,6 +52,7 @@ const outcome = (over: Partial<ShardOutcome>): ShardOutcome => ({
elapsedMs: 1000,
groupPid: null,
executedTests: 3,
skippedTests: null,
...over,
});
@@ -142,7 +149,8 @@ describe('recorded-duration slice packing', () => {
test('the PR-profile file set spreads recorded time instead of stacking it', () => {
const env = { EVALS_ALL: '1' };
const discovered = Object.keys(recorded);
// Seed files only: case-shard keys (`<file>#<case>`) come from expansion.
const discovered = Object.keys(recorded).filter(key => !key.includes('#'));
const load = (files: string[]) => files.reduce((sum, file) => sum + recorded[file], 0);
const packed = lanes(buildRunManifest({ tier: 'gate', sliceCount: 6, evalsAll: true, env, discovered })).map(load);
const baseline = lanes(buildRunManifest({ tier: 'gate', sliceCount: 6, evalsAll: true, env, discovered, durations: {} })).map(load);
@@ -171,6 +179,88 @@ describe('recorded-duration slice packing', () => {
});
});
describe('budget slice packing', () => {
const s = (seconds: number) => seconds * 1000;
test('best-fit packs recorded work under the budget; unknown and over-budget work gets its own runner', () => {
const recorded = { 'test/a.test.ts': s(500), 'test/b.test.ts': s(300), 'test/c.test.ts': s(240), 'test/d.test.ts': s(100), 'test/long.test.ts': s(900) };
const files = [...Object.keys(recorded), 'test/unknown.test.ts'];
const plan = packBySliceBudget(files, s(540), 2, recorded);
expect(plan.slices.flat().sort()).toEqual([...files].sort());
expect(plan.estimates['test/unknown.test.ts']).toBe(s(540));
for (const [index, slice] of plan.slices.entries()) {
expect(plan.estimatedSliceMs[index]).toBe(estimatedSliceMs(slice, file => plan.estimates[file]!, 2));
if (slice.length > 1) expect(plan.estimatedSliceMs[index]).toBeLessThanOrEqual(s(540));
}
expect(plan.slices).toContainEqual(['test/long.test.ts']);
expect(plan.slices.find(slice => slice.includes('test/unknown.test.ts'))!.length).toBeLessThanOrEqual(2);
// Deterministic regardless of discovery order.
expect(packBySliceBudget([...files].reverse(), s(540), 2, recorded)).toEqual(plan);
const worst = Math.max(...plan.slices.map(slice => sliceSupervisedWallMs(slice, 2)));
expect(plan.ciTimeoutMinutes).toBe(Math.ceil(worst / 60_000) + CI_SETUP_ALLOWANCE_MINUTES);
expect(packBySliceBudget([], s(540), 2, recorded)).toMatchObject({ slices: [[]], ciTimeoutMinutes: CI_SETUP_ALLOWANCE_MINUTES });
});
test('overlays keep one final slice at their one-at-a-time admission', () => {
const overlays = ['test/skill-e2e-overlay-harness-a.test.ts', 'test/skill-e2e-overlay-harness-b.test.ts'];
const plan = packBySliceBudget(['test/a.test.ts', ...overlays], s(540), 2, { [overlays[0]!]: s(100), [overlays[1]!]: s(200), 'test/a.test.ts': s(10) });
expect(plan.slices.at(-1)).toEqual([overlays[1], overlays[0]]);
expect(plan.estimatedSliceMs.at(-1)).toBe(s(300));
});
test('live census plans: multi-file slices stay within the budget and the executor runs longest first', () => {
for (const tier of ['gate', 'periodic'] as const) {
const manifest = buildRunManifest({ tier, sliceBudgetMs: s(540), jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } });
expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest);
for (let slice = 1; slice <= manifest.sliceCount; slice++) {
const entries = sliceExecutionOrder(manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice));
const estimate = manifest.plan!.estimatedSliceMs[slice - 1]!;
if (entries.length > 1 && !entries.some(entry => entry.file.includes('overlay-harness'))) expect(estimate, `${tier} slice ${slice}`).toBeLessThanOrEqual(s(540));
expect(entries.map(entry => entry.estimatedMs)).toEqual([...entries.map(entry => entry.estimatedMs!)].sort((a, b) => b - a));
}
}
});
test('plans fail closed on malformed metadata, mixed modes, and a worker count the plan did not supervise', () => {
const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: s(540), jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } });
for (const broken of [
{ ...manifest, plan: { ...manifest.plan!, estimatedSliceMs: [] } },
{ ...manifest, plan: { ...manifest.plan!, jobs: 0 } },
{ ...manifest, entries: manifest.entries.map(entry => ({ ...entry, estimatedMs: undefined })) },
]) expect(() => parseRunManifest(JSON.stringify(broken))).toThrow('slice plan malformed');
expect(() => buildRunManifest({ tier: 'gate', sliceCount: 2, sliceBudgetMs: s(540), jobs: 2, evalsAll: true })).toThrow('exactly one');
expect(() => buildRunManifest({ tier: 'gate', sliceBudgetMs: s(540), evalsAll: true })).toThrow('explicit positive --jobs');
expect(() => parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540'], {})).toThrow('explicit --jobs');
expect(() => parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540', '--jobs', '2', '--slices', '3'], {})).toThrow('exactly one');
expect(parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540'], { EVALS_JOBS: '2' }).sliceBudgetMs).toBe(s(540));
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-plan-jobs-'));
try {
const planPath = path.join(dir, 'manifest.json');
fs.writeFileSync(planPath, JSON.stringify(manifest));
const result = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--plan', planPath, '--slice', '1', '--list', '--jobs', '1'],
{ cwd: ROOT, encoding: 'utf8', timeout: 10_000, env: { PATH: path.dirname(process.execPath), HOME: dir, EVALS_TIER: 'gate' } });
expect(result.status).toBe(1);
expect(result.stderr).toContain('manifest was packed for 2 worker(s) per slice');
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
});
test('the duration seed is per tier and a report rewrite keeps the other tiers', () => {
const gate = loadPaidTestDurations(ROOT, 'gate');
const periodic = loadPaidTestDurations(ROOT, 'periodic');
expect(gate['test/skill-e2e-plan.test.ts']).not.toBe(periodic['test/skill-e2e-plan.test.ts']);
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-durations-'));
try {
fs.mkdirSync(path.join(dir, 'scripts'));
writePaidTestDurations('periodic', { 'test/p.test.ts': 2_000 }, dir);
writePaidTestDurations('gate', { 'test/g.test.ts': 3_000 }, dir);
expect(loadPaidTestDurations(dir, 'gate')).toEqual({ 'test/g.test.ts': 3_000 });
expect(loadPaidTestDurations(dir, 'periodic')).toEqual({ 'test/p.test.ts': 2_000 });
fs.writeFileSync(path.join(dir, 'scripts/paid-test-durations.json'), JSON.stringify({ version: 1, durations: { 'test/g.test.ts': 1 } }));
expect(loadPaidTestDurations(dir, 'gate')).toEqual({});
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
});
});
describe('manifest executor scope', () => {
test('list-only validates and prints the selected manifest slice without launching tests or writing results', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-manifest-list-'));
@@ -184,8 +274,8 @@ test('local launch sentinel', () => writeFileSync(${JSON.stringify(receipt)}, 't
version: 1, tier: 'gate', evalsAll: true, sliceCount: 3, selectionReason: 'local list-only fixture',
entries: [
{ file, slice: 1, status: 'planned' },
{ file: 'test/skill-e2e-plan.test.ts', slice: 2, status: 'planned',
budget: resolvePaidShardBudget(['test/skill-e2e-plan.test.ts']) },
{ file: 'test/skill-e2e-plan.test.ts#plan-ceo-review', slice: 2, status: 'planned',
budget: resolvePaidShardBudget(['test/skill-e2e-plan.test.ts#plan-ceo-review']) },
],
};
const manifestPath = path.join(dir, 'manifest.json');
@@ -327,7 +417,7 @@ describe('slice-result reconciliation (report)', () => {
tier: 'gate',
sliceIndex: index,
sliceCount: 2,
outcomes: files.map((f) => ({ files: [f], status, exitCode: 0, elapsedMs: 5, executedTests: 2 })),
outcomes: files.map((f) => ({ files: [f], status, exitCode: 0, elapsedMs: 5, executedTests: 2, skippedTests: 0 })),
});
test('all slices present and passing → ok', () => {
@@ -397,25 +487,24 @@ describe('hollow-shard guard', () => {
});
describe('retry parity', () => {
test('registered native workflows preserve main retry policy while overlay attempts stay isolated', () => {
test('registered native workflows follow the retry rule while overlay attempts stay isolated', () => {
// A 25-minute case is past RETRY_MAX_CASE_MS: its timed-out attempt is the verdict.
const native = 'test/skill-e2e-plan-ceo-split-overflow.test.ts';
expect(retriesForFiles([native])).toBe(1);
expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(1);
expect(buildPaidShardArgs([native], 1_800_000, 2, retriesForFiles([native])).join(' ')).toContain('--retry 1');
expect(retriesForFiles([native])).toBe(0);
expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(0);
expect(buildPaidShardArgs([native], 1_800_000, 2, retriesForFiles([native])).join(' ')).toContain('--retry 0');
const overlay = 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts';
expect(retriesForFiles([overlay])).toBe(0);
});
test('overrides exist only for the files whose matrix rows earned them, and each names a real file', () => {
expect(Object.keys(RETRY_OVERRIDES).sort()).toEqual([
'test/skill-e2e-office-hours-auto-mode.test.ts',
'test/skill-e2e-plan-mode-no-op.test.ts',
'test/skill-e2e-workflow.test.ts',
]);
for (const file of Object.keys(RETRY_OVERRIDES)) {
expect(fs.existsSync(path.join(ROOT, file)), `stale RETRY_OVERRIDES entry: ${file}`).toBe(true);
test('the matrix-era earned retries now follow the timeout-is-a-verdict rule, and each names a real file', () => {
// These three old matrix rows earned `retries: 2`; every one has a
// CAPTURE_LONG case, so a timed-out attempt is now their verdict.
for (const file of ['test/skill-e2e-office-hours-auto-mode.test.ts', 'test/skill-e2e-plan-mode-no-op.test.ts', 'test/skill-e2e-workflow.test.ts']) {
expect(fs.existsSync(path.join(ROOT, file)), `stale retry parity entry: ${file}`).toBe(true);
expect(retriesForFiles([file])).toBe(0);
}
expect(retriesForFiles(['test/skill-e2e-workflow.test.ts'])).toBe(2);
expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(1);
expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(0);
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1);
expect(buildPaidShardArgs(['x'], 1000, 4, 2)).toContain('2');
expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 1');
});
+114 -1
View File
@@ -13,6 +13,7 @@ import { describe, test, expect } from 'bun:test';
import * as fs from 'fs';
import * as os from 'os';
import * as path from 'path';
import { E2E_TIERS, E2E_TOUCHFILES } from './helpers/touchfiles';
const ROOT = path.resolve(import.meta.dir, '..');
import {
@@ -31,7 +32,21 @@ import {
summarize,
summaryExitCode,
tierSkipReason,
marathonSkipReason,
CASE_SHARDED_FILES,
CASE_TEST_NAMES,
caseTestNamePattern,
expandCaseShards,
fileCaseRegistration,
resolvePaidShardBudget,
retriesForFiles,
shardCaseId,
shardFile,
shardSlug,
verifySliceResults,
selectPaidTestFiles,
buildRunManifest,
parseRunManifest,
type ShardOutcome,
} from '../scripts/test-paid-shards';
@@ -117,7 +132,105 @@ describe('tier lane skip (B5)', () => {
}
const workflow = fs.readFileSync(path.join(ROOT, '.github/workflows/evals-periodic.yml'), 'utf8');
expect(workflow.match(/--skip-judges/g)).toHaveLength(1);
expect(workflow).toMatch(/--tier gate --emit-plan \/tmp\/gate-census-plan\/manifest\.json --slices 7 --skip-judges/);
expect(workflow).toMatch(/--tier gate --emit-plan \/tmp\/gate-census-plan\/manifest\.json --slice-budget 540 --jobs 2 --skip-judges/);
});
});
describe('marathon tier lane', () => {
const file = 'test/skill-e2e-sample.test.ts';
const reg = { 'sample-gate': [file], 'sample-long': [file] } as Record<string, string[]>;
const tiers = { 'sample-gate': 'gate', 'sample-long': 'marathon' };
test('a marathon-declared file never enters the gate or periodic lane', () => {
const source = "const describeE2E = describeE2ETier('marathon');";
expect(classifyPaidTestFile(source, 'marathon')).toEqual({ included: true, reason: "declares tier 'marathon'" });
for (const tier of ['gate', 'periodic'] as const) {
expect(classifyPaidTestFile(source, tier)).toEqual({ included: false, reason: "declares tier 'marathon' only" });
}
expect(marathonSkipReason(file, source, {}, {})).toBeNull();
});
test('marathon selects positively: only declared files or files registering a marathon case', () => {
expect(marathonSkipReason(file, "testIfSelected('sample-long', async () => {});", reg, tiers)).toBeNull();
expect(marathonSkipReason(file, "testIfSelected(name, async () => {});", reg, tiers)).toBeNull();
const gateOnly = { 'sample-gate': [file] };
for (const source of ["testIfSelected('sample-gate', async () => {});", "testIfSelected(name, async () => {});", ''])
expect(marathonSkipReason(file, source, gateOnly, tiers)).toBe('skipped: declares no marathon tier and registers no marathon case');
const periodic = "const describeE2E = describeE2ETier('periodic');";
expect(classifyPaidTestFile(periodic, 'marathon')).toEqual({ included: false, reason: "declares tier 'periodic' only" });
});
test('a registered marathon case keeps its gate sibling scheduled in the gate lane', () => {
const source = "testIfSelected('sample-gate', async () => {}); testIfSelected('sample-long', async () => {});";
expect(tierSkipReason(file, source, 'gate', reg, tiers)).toBeNull();
expect(tierSkipReason(file, source, 'periodic', reg, tiers)).toBe('skipped: no E2E_TIERS id has tier periodic');
});
test('the live marathon lane only plans files that carry marathon work, never the LLM judges', () => {
const { selected, excluded } = selectPaidTestFiles(collectPaidTestFiles(), 'marathon');
expect(selected).not.toContain('test/skill-llm-eval.test.ts');
for (const file of selected) {
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
expect(marathonSkipReason(file, source), file).toBeNull();
}
expect(selected.length + excluded.length).toBe(collectPaidTestFiles().length);
});
});
describe('case-sharded files', () => {
// Paid cases only: `if (!evalsEnabled) test(...)` blocks are free checks.
const caseNames = (source: string) => [...source.matchAll(
/(?<![.\w])(?<!if \(!evalsEnabled\) )(?:testConcurrentIfSelected|testIfSelected|test(?:\.serial|\.concurrent)?)\(\s*(['"])(.+?)\1/g,
)].map(match => match[2]!);
for (const file of CASE_SHARDED_FILES) {
test(`${file}: every Bun case is a registered E2E case, so every case gets a shard`, () => {
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
const { registered, known } = fileCaseRegistration(file, source);
expect(known).toBe(true);
expect(caseNames(source).sort()).toEqual(registered.map(id => CASE_TEST_NAMES[id] ?? id).sort());
const keys = (['gate', 'periodic', 'marathon'] as const).flatMap(tier => expandCaseShards([file], tier));
expect(keys.map(key => shardCaseId(key)).sort()).toEqual([...registered].sort());
for (const key of keys) expect(shardFile(key)).toBe(file);
});
}
test('a case key runs exactly its case: exact name pattern, own eval slug, per-case supervision', () => {
const pattern = new RegExp(caseTestNamePattern(['design-review-detector-shim']));
expect(pattern.test('Design review detector shim E2E design-review-detector-shim')).toBe(true);
expect(pattern.test('Design review detector shim E2E design-review-detector-shim-dom')).toBe(false);
expect(new RegExp(caseTestNamePattern(['plan-review-report'])).test('Plan Review Report E2E /plan-eng-review writes GSTACK REVIEW REPORT to plan file')).toBe(true);
const key = 'test/skill-e2e-plan.test.ts#plan-ceo-review';
expect(shardSlug([key])).toBe('skill-e2e-plan--plan-ceo-review');
expect(shardSlug([key])).not.toBe(shardSlug(['test/skill-e2e-plan.test.ts#plan-eng-review']));
expect(retriesForFiles([key])).toBe(retriesForFiles(['test/skill-e2e-plan.test.ts']));
const whole = resolvePaidShardBudget(['test/skill-e2e-plan.test.ts']);
const one = resolvePaidShardBudget([key]);
expect(one.policyId).toBe(whole.policyId);
expect(one.timeoutMs).toBeLessThan(whole.timeoutMs);
expect(planPaidShards([key, 'test/skill-e2e-plan.test.ts#plan-eng-review', 'test/a.test.ts'], { maxFilesPerShard: 3 }))
.toEqual([['test/a.test.ts'], [key], ['test/skill-e2e-plan.test.ts#plan-eng-review']]);
});
test('manifests plan each case once and results must execute exactly that case', () => {
const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: 540_000, jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } });
const planned = manifest.entries.filter(entry => entry.status === 'planned');
for (const file of CASE_SHARDED_FILES) {
expect(planned.some(entry => entry.file === file)).toBe(false);
const gateIds = Object.keys(E2E_TOUCHFILES).filter(id => E2E_TOUCHFILES[id]!.includes(file) && E2E_TIERS[id] === 'gate');
expect(planned.filter(entry => shardFile(entry.file) === file).map(entry => shardCaseId(entry.file)).sort()).toEqual(gateIds.sort());
}
const results = Array.from({ length: manifest.sliceCount }, (_, i) => ({ version: 1 as const, tier: 'gate' as const, sliceIndex: i + 1, sliceCount: manifest.sliceCount,
outcomes: planned.filter(entry => entry.slice === i + 1).map(entry => ({ files: [entry.file], status: 'passed' as const, exitCode: 0, elapsedMs: 1,
executedTests: shardCaseId(entry.file) ? 3 : 1, skippedTests: shardCaseId(entry.file) ? 2 : 0, ...(entry.budget ? { budget: entry.budget } : {}) })) }));
expect(verifySliceResults(manifest, results).problems.filter(problem => problem.includes('Case shard'))).toEqual([]);
const empty = structuredClone(results);
const victim = empty.flatMap(result => result.outcomes).find(outcome => shardCaseId(outcome.files[0]!))!;
victim.skippedTests = victim.executedTests;
expect(verifySliceResults(manifest, empty).problems).toContain(`Case shard must execute exactly its one case: ${victim.files[0]}`);
const whole: any = structuredClone(manifest);
whole.entries.push({ file: CASE_SHARDED_FILES[0], slice: 1, status: 'planned', estimatedMs: 1 });
expect(() => parseRunManifest(JSON.stringify(whole))).toThrow('one registered case per shard');
});
});