mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
feat(evals): ~12-minute blocking paid lanes and a non-blocking marathon lane
- Planner budget mode (--slice-budget S --jobs J): recorded per-tier wall times pack into as many ~9-minute executors as the work needs; the plan records per-slice estimates and the CI job timeout (supervised worst case + 20 min). evals.yml and evals-periodic.yml derive matrix size and timeout-minutes from it; max-parallel covers every slice at once. - Case shards: plan/design/review-army/shared-libs(-paths) run one registered case per process (<file>#<case id>, exact name pattern, exactly one case). - Retry rule: a timed-out attempt is a verdict. Only files whose every case budget is CAPTURE tier or shorter keep one retry; walls shrink to match. - Marathon tier: positive selection, excluded from gate/periodic planners, run by the new evals-marathon.yml (weekly + dispatch, fresh, own report). - PR-lane E2E reuse of verified first-attempt passes on identical inputs; the report rejects reuse outside the fast PR profile. - Duration seed from census run 36385945043, per tier and per case shard.
This commit is contained in:
1 parent
acb7bc02b1
commit
0023d011a3
18 files changed
+1773
-450
No files matched your search
@@ -0,0 +1,298 @@
|
||||
name: Marathon Evals
|
||||
# The NON-BLOCKING marathon lane: complete start-to-finish flows (tier
|
||||
# 'marathon' in test/helpers/touchfiles-data.ts / describeE2ETier('marathon'))
|
||||
# that take longer than a blocking lane's ~12-minute wall. They never run in
|
||||
# the PR gate (evals.yml) or the weekly periodic + gate census
|
||||
# (evals-periodic.yml); nothing requires this workflow, so a red marathon
|
||||
# reports through its own tracking issue without gating any merge. Same engine
|
||||
# and FAIL-CLOSED report as the other lanes: one planner manifest, one file per
|
||||
# runner, a missing slice artifact is a failure. Always fresh: no result reuse.
|
||||
on:
|
||||
schedule:
|
||||
- cron: '0 12 * * 6' # Saturday 12:00 UTC, clear of the Monday periodic census
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
group: evals-marathon
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
IMAGE: ghcr.io/${{ github.repository }}/ci
|
||||
EVALS_PROFILE: full
|
||||
EVALS_FRESH: "1"
|
||||
EVALS_CACHE_PURPOSE: marathon
|
||||
|
||||
jobs:
|
||||
IMAGE: ghcr.io/${{ github.repository }}/ci
|
||||
EVALS_PROFILE: full
|
||||
EVALS_FRESH: "1"
|
||||
EVALS_CACHE_PURPOSE: periodic
|
||||
|
||||
jobs:
|
||||
build-image:
|
||||
runs-on: ubicloud-standard-8
|
||||
timeout-minutes: 15
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
outputs:
|
||||
image-tag: ${{ steps.meta.outputs.tag }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
|
||||
- id: meta
|
||||
# Keep in sync with evals.yml and evals-periodic.yml — key on Dockerfile + lockfile only
|
||||
# (package.json's version field would bust the key on every ship).
|
||||
# Byte-identity pinned by test/ci-image-tag-binding.test.ts.
|
||||
run: echo "tag=${{ env.IMAGE }}:${{ hashFiles('.github/docker/Dockerfile.ci', 'bun.lock', 'patches/**') }}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Check if image exists
|
||||
id: check
|
||||
run: |
|
||||
if docker manifest inspect ${{ steps.meta.outputs.tag }} > /dev/null 2>&1; then
|
||||
echo "exists=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "exists=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- if: steps.check.outputs.exists == 'false'
|
||||
run: cp package.json bun.lock .github/docker/ && cp -R patches .github/docker/patches
|
||||
|
||||
# Registry cache export needs a docker-container builder — the default
|
||||
# `docker` driver hard-errors on cache-to.
|
||||
- if: steps.check.outputs.exists == 'false'
|
||||
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4
|
||||
|
||||
- if: steps.check.outputs.exists == 'false'
|
||||
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7
|
||||
with:
|
||||
context: .github/docker
|
||||
file: .github/docker/Dockerfile.ci
|
||||
push: true
|
||||
# Cron-triggered in the base repo only, so cache export is always safe here.
|
||||
cache-from: type=registry,ref=${{ env.IMAGE }}:buildcache
|
||||
cache-to: type=registry,ref=${{ env.IMAGE }}:buildcache,mode=max
|
||||
tags: |
|
||||
${{ steps.meta.outputs.tag }}
|
||||
${{ env.IMAGE }}:latest
|
||||
|
||||
|
||||
plan-slices:
|
||||
runs-on: ubicloud-standard-8
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
slices: ${{ steps.matrix.outputs.slices }}
|
||||
timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
|
||||
with:
|
||||
bun-version: 1.4.0
|
||||
|
||||
# One marathon file per runner: a 1-second budget never packs two
|
||||
# recorded files together.
|
||||
- name: Emit run manifest (ALL marathon tests)
|
||||
env:
|
||||
EVALS_ALL: "1"
|
||||
run: EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --emit-plan /tmp/marathon-plan/manifest.json --slice-budget 1 --jobs 1
|
||||
|
||||
- name: Derive the executor matrix from the plan
|
||||
id: matrix
|
||||
run: |
|
||||
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: marathon-plan
|
||||
path: /tmp/marathon-plan/manifest.json
|
||||
retention-days: 30
|
||||
|
||||
eval-slices:
|
||||
runs-on: ubicloud-standard-8
|
||||
needs: [build-image, plan-slices]
|
||||
env:
|
||||
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-marathon-${{ matrix.slice }}
|
||||
# One marathon file per runner; the job timeout is the plan's supervised
|
||||
# worst case plus 20 minutes setup/upload.
|
||||
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
container:
|
||||
image: ${{ needs.build-image.outputs.image-tag }}
|
||||
credentials:
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
options: --user runner
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 8
|
||||
matrix:
|
||||
slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
# Full history: files with SELF-derived selection (the LLM-judge
|
||||
# map, routing) walk git at module load, and selection is
|
||||
# fail-closed on git errors — a shallow checkout crashed those
|
||||
# shards on the lane's first live run ("ambiguous argument
|
||||
# 'main...HEAD'"). The manifest still governs WHICH shards run.
|
||||
fetch-depth: 0
|
||||
persist-credentials: false
|
||||
|
||||
- name: Fix bun temp
|
||||
uses: ./.github/actions/fix-bun-temp
|
||||
|
||||
- name: Restore deps
|
||||
uses: ./.github/actions/restore-deps
|
||||
|
||||
- run: bun run build
|
||||
|
||||
# Any slice can host a PTY test — seed + registration run
|
||||
# unconditionally (idempotent; mirrors evals.yml's sliced lane). The
|
||||
# register composite carries the fail-fast dangling-symlink/frontmatter
|
||||
# verification loop — this lane previously LACKED it, so a moved skill
|
||||
# target surfaced as a silent "Unknown command" + wedged PTY session.
|
||||
- name: Seed claude interactive config
|
||||
uses: ./.github/actions/seed-claude-config
|
||||
with:
|
||||
anthropic-api-key: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
|
||||
- name: Register gstack skills for PTY tests
|
||||
uses: ./.github/actions/register-gstack-skills
|
||||
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
||||
with:
|
||||
name: marathon-plan
|
||||
path: /tmp/marathon-plan
|
||||
|
||||
- name: Run marathon slice ${{ matrix.slice }}
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
||||
PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers
|
||||
EVALS_JOBS: "1"
|
||||
EVALS_CONCURRENCY: "2"
|
||||
GSTACK_EVAL_DIR: /tmp/marathon-slice-results
|
||||
run: EVALS_TIER=marathon bun run scripts/test-paid-shards.ts --tier marathon --plan /tmp/marathon-plan/manifest.json --slice ${{ matrix.slice }}
|
||||
|
||||
- name: Upload slice results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: marathon-slice-${{ matrix.slice }}
|
||||
path: /tmp/marathon-slice-results
|
||||
retention-days: 90
|
||||
|
||||
- name: Upload native capture evidence
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: native-captures-${{ env.EVALS_RUN_ID }}
|
||||
include-hidden-files: true
|
||||
path: |
|
||||
~/.gstack/projects/*/e2e-runs
|
||||
~/.gstack/projects/*/evals/qa-callers
|
||||
~/.gstack-dev/e2e-runs
|
||||
~/.gstack-dev/evals/qa-callers
|
||||
if-no-files-found: ignore
|
||||
retention-days: 90
|
||||
|
||||
- name: Upload shard logs on failure
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: marathon-slice-${{ matrix.slice }}-logs
|
||||
include-hidden-files: true
|
||||
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
|
||||
# runner's spool lands THERE, not /tmp — the original /tmp glob
|
||||
# uploaded nothing and a red slice's diagnostics were unreachable.
|
||||
path: |
|
||||
/home/runner/.cache/gstack-paid-shard-*.log
|
||||
/tmp/gstack-paid-shard-*.log
|
||||
if-no-files-found: ignore
|
||||
retention-days: 30
|
||||
|
||||
report:
|
||||
runs-on: ubicloud-standard-2
|
||||
needs: [plan-slices, eval-slices]
|
||||
# !cancelled(): the report must run (and FAIL) when an executor died — a
|
||||
# missing slice artifact reading as green is the class this lane kills —
|
||||
# but a cancelled run stops here.
|
||||
if: ${{ !cancelled() && needs.plan-slices.result == 'success' }}
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
issues: write
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
|
||||
with:
|
||||
bun-version: 1.4.0
|
||||
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
||||
with:
|
||||
name: marathon-plan
|
||||
path: /tmp/marathon-report
|
||||
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
||||
with:
|
||||
pattern: marathon-slice-[0-9]*
|
||||
path: /tmp/marathon-report
|
||||
merge-multiple: true
|
||||
|
||||
- name: Reconcile slices against the manifest (fail-closed)
|
||||
id: reconcile
|
||||
if: always()
|
||||
run: |
|
||||
set +e
|
||||
EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --report /tmp/marathon-report | tee /tmp/report.txt
|
||||
# PIPESTATUS[0], NOT $?: the default step shell has no pipefail.
|
||||
echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# One tracking issue for the whole lane (never one per week).
|
||||
- name: Upsert tracking issue on failure
|
||||
if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success')
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
TITLE="Weekly marathon evals: red lane needs triage"
|
||||
BODY_FILE=/tmp/issue-body.md
|
||||
{
|
||||
echo "Automated weekly marathon report (non-blocking lane) — run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
|
||||
echo
|
||||
echo "- reconciliation exit: ${{ steps.reconcile.outputs.exit }}"
|
||||
echo "- marathon slices job: ${{ needs.eval-slices.result }}"
|
||||
echo
|
||||
echo '```'
|
||||
tail -c 6000 /tmp/report.txt 2>/dev/null || echo "(no reconciliation output)"
|
||||
echo '```'
|
||||
} > "$BODY_FILE"
|
||||
EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty')
|
||||
if [ -n "$EXISTING" ]; then
|
||||
gh issue comment "$EXISTING" --repo "$GITHUB_REPOSITORY" --body-file "$BODY_FILE"
|
||||
echo "commented on #$EXISTING"
|
||||
else
|
||||
gh issue create --repo "$GITHUB_REPOSITORY" --title "$TITLE" --body-file "$BODY_FILE"
|
||||
fi
|
||||
|
||||
- name: Fail the workflow when reconciliation failed
|
||||
if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success')
|
||||
run: exit 1
|
||||
@@ -4,8 +4,13 @@ name: Periodic Evals
|
||||
# tests can't rot invisibly — the class where the autoplan-dual-voice E2E was
|
||||
# silently broken for months until a lucky local diff selected it. Engine:
|
||||
# scripts/test-paid-shards.ts (the same runner local eval:bg:periodic uses):
|
||||
# one planner manifest, 6 ordinary slices plus an overlay slice, and a FAIL-CLOSED report — a slice
|
||||
# whose artifact never landed is a failure, not an absence. The gate-census
|
||||
# one planner manifest packed by recorded durations into as many ~9-minute
|
||||
# executors as the work needs (one file, or a tightly packed group, per
|
||||
# runner; overlays share one final slice), and a FAIL-CLOSED report — a slice
|
||||
# whose artifact never landed is a failure, not an absence. The matrix size
|
||||
# and job timeout come from the plan, so they cannot drift from the census.
|
||||
# Full end-to-end flows run in the non-blocking marathon lane
|
||||
# (evals-marathon.yml), never here. The gate-census
|
||||
# job is the weekly EVALS_ALL backstop for the gate tier (PR lanes are
|
||||
# diff-billed, so without it the full gate census might never execute
|
||||
# anywhere); the hollow-shard guard (exit 0 + zero executed tests under
|
||||
@@ -84,6 +89,11 @@ jobs:
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
periodic_slices: ${{ steps.periodic-matrix.outputs.slices }}
|
||||
periodic_timeout_minutes: ${{ steps.periodic-matrix.outputs.timeout_minutes }}
|
||||
gate_slices: ${{ steps.gate-matrix.outputs.slices }}
|
||||
gate_timeout_minutes: ${{ steps.gate-matrix.outputs.timeout_minutes }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -96,7 +106,13 @@ jobs:
|
||||
- name: Emit run manifest (ALL periodic tests minus reasoned excludes)
|
||||
env:
|
||||
EVALS_ALL: "1"
|
||||
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 7
|
||||
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2
|
||||
|
||||
- name: Derive the periodic executor matrix from the plan
|
||||
id: periodic-matrix
|
||||
run: |
|
||||
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
@@ -107,7 +123,13 @@ jobs:
|
||||
- name: Emit gate census manifest (ALL gate tests)
|
||||
env:
|
||||
EVALS_ALL: "1"
|
||||
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 7 --skip-judges
|
||||
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges
|
||||
|
||||
- name: Derive the gate census executor matrix from the plan
|
||||
id: gate-matrix
|
||||
run: |
|
||||
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/gate-census-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/gate-census-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
@@ -120,9 +142,10 @@ jobs:
|
||||
needs: [build-image, plan-slices]
|
||||
env:
|
||||
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }}
|
||||
# Seven slices retain every registered case and retry. The complete
|
||||
# census needs at most 244m40s per slice, plus 20 minutes setup/upload.
|
||||
timeout-minutes: 360
|
||||
# The planner packs ~9 minutes of recorded work per slice; the job timeout
|
||||
# is its supervised worst case (every shard at its wall) plus 20 minutes
|
||||
# setup/upload, computed from the same manifest the slices execute.
|
||||
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
@@ -134,9 +157,11 @@ jobs:
|
||||
options: --user runner
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 8
|
||||
# Every planned slice starts at once; test/evals-workflow-wiring.test.ts
|
||||
# fails when the live plan outgrows this cap.
|
||||
max-parallel: 24
|
||||
matrix:
|
||||
slice: [1, 2, 3, 4, 5, 6, 7]
|
||||
slice: ${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -174,7 +199,7 @@ jobs:
|
||||
name: paid-plan
|
||||
path: /tmp/paid-plan
|
||||
|
||||
- name: Run slice ${{ matrix.slice }}/7
|
||||
- name: Run periodic slice ${{ matrix.slice }}
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
@@ -231,8 +256,8 @@ jobs:
|
||||
needs: [build-image, plan-slices]
|
||||
env:
|
||||
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-gate-census-${{ matrix.slice }}
|
||||
# Seven slices need at most 272m each, plus 20 minutes setup/upload.
|
||||
timeout-minutes: 352
|
||||
# Supervised worst case of the packed plan plus 20 minutes setup/upload.
|
||||
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.gate_timeout_minutes) }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
@@ -243,11 +268,11 @@ jobs:
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
options: --user runner
|
||||
strategy:
|
||||
# Four file workers total, each retaining two in-file case workers.
|
||||
# Two file workers per slice, each retaining two in-file case workers.
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
max-parallel: 16
|
||||
matrix:
|
||||
slice: [1, 2, 3, 4, 5, 6, 7]
|
||||
slice: ${{ fromJSON(needs.plan-slices.outputs.gate_slices) }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -272,13 +297,13 @@ jobs:
|
||||
name: gate-census-plan
|
||||
path: /tmp/gate-census-plan
|
||||
|
||||
- name: Run gate census slice ${{ matrix.slice }}/7
|
||||
- name: Run gate census slice ${{ matrix.slice }}
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
||||
PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers
|
||||
EVALS_JOBS: "1"
|
||||
EVALS_JOBS: "2"
|
||||
EVALS_CONCURRENCY: "2"
|
||||
GSTACK_EVAL_DIR: /tmp/gate-census-results
|
||||
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/gate-census-plan/manifest.json --slice ${{ matrix.slice }}
|
||||
|
||||
+29
-22
@@ -125,6 +125,9 @@ jobs:
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
slices: ${{ steps.matrix.outputs.slices }}
|
||||
timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -141,7 +144,7 @@ jobs:
|
||||
if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all'
|
||||
env:
|
||||
EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }}
|
||||
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slices 7
|
||||
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2
|
||||
|
||||
- name: Emit validation-phase manifest
|
||||
if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all'
|
||||
@@ -152,23 +155,30 @@ jobs:
|
||||
run: |
|
||||
bun --no-install -e '
|
||||
import { mkdirSync, writeFileSync } from "node:fs";
|
||||
import { buildRunManifest, collectPaidTestFiles } from "./scripts/test-paid-shards.ts";
|
||||
import { buildRunManifest, collectPaidTestFiles, restrictManifestSelection } from "./scripts/test-paid-shards.ts";
|
||||
const phase = process.env.VALIDATION_PHASE;
|
||||
if (!["quality", "cookie-quality", "behavior", "cookie-behavior"].includes(phase)) throw new Error("Invalid validation phase");
|
||||
const cookieBehavior = phase === "cookie-behavior";
|
||||
const discovered = phase === "cookie-quality" ? ["test/skill-llm-eval.test.ts"]
|
||||
: cookieBehavior ? ["test/skill-e2e-bws.test.ts", "test/skill-e2e-qa-workflow.test.ts", "test/skill-e2e-design.test.ts", "test/skill-e2e-diagram.test.ts", "test/skill-e2e-deploy.test.ts"]
|
||||
: collectPaidTestFiles().filter(file => file.startsWith("test/skill-llm-eval") === (phase === "quality"));
|
||||
const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceCount: 6, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered,
|
||||
const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceBudgetMs: 540000, jobs: 2, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered,
|
||||
...(cookieBehavior ? { changedFiles: ["browse/src/cookie-picker-routes.ts", "browse/src/cookie-import-browser.ts", "browse/src/bun-polyfill.cjs"], env: { ...process.env, EVALS_ALL: "" } } : {}) });
|
||||
if (phase === "cookie-quality") manifest.selection = { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] };
|
||||
if (cookieBehavior) manifest.selection = { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] };
|
||||
manifest.selectionReason = phase + " validation subset; " + manifest.selectionReason;
|
||||
const subset = phase === "cookie-quality" ? { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] }
|
||||
: cookieBehavior ? { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] } : null;
|
||||
const restricted = subset ? restrictManifestSelection(manifest, subset, "outside the " + phase + " validation subset") : manifest;
|
||||
restricted.selectionReason = phase + " validation subset; " + manifest.selectionReason;
|
||||
mkdirSync("/tmp/paid-plan", { recursive: true });
|
||||
writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(manifest, null, 2) + "\n");
|
||||
console.log(phase + ": " + manifest.entries.filter(entry => entry.status === "planned").length + " planned shards");
|
||||
writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(restricted, null, 2) + "\n");
|
||||
console.log(phase + ": " + restricted.entries.filter(entry => entry.status === "planned").length + " planned shards");
|
||||
'
|
||||
|
||||
- name: Derive the executor matrix from the plan
|
||||
id: matrix
|
||||
run: |
|
||||
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: paid-plan
|
||||
@@ -184,14 +194,11 @@ jobs:
|
||||
# (image already published), but a newer push's cancel-in-progress stops
|
||||
# it instead of letting a superseded run finish its paid slices first.
|
||||
if: ${{ !cancelled() && needs.build-image.result == 'success' && needs.plan-slices.result == 'success' }}
|
||||
# Aggregate spawn-concurrency budget: 6 slices x EVALS_JOBS=2 x
|
||||
# EVALS_CONCURRENCY=2 = 24 concurrent tests lane-wide (the old matrix's
|
||||
# 40-way per row queued claude session STARTUP behind 39 siblings and ate
|
||||
# per-test budgets — the documented timeout-flake family). Tune with
|
||||
# parity data before raising.
|
||||
# The complete gate census needs at most 242 minutes per slice; keep
|
||||
# 20 minutes for setup/upload without preempting configured retries.
|
||||
timeout-minutes: 265
|
||||
# The planner packs ~9 minutes of recorded work per slice (EVALS_JOBS=2 x
|
||||
# EVALS_CONCURRENCY=2 per runner, never the old 40-way per-row fan-out
|
||||
# that queued claude session STARTUP behind 39 siblings). The job timeout
|
||||
# is the plan's supervised worst case plus 20 minutes setup/upload.
|
||||
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
@@ -203,9 +210,9 @@ jobs:
|
||||
options: --user runner
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 6
|
||||
max-parallel: 16
|
||||
matrix:
|
||||
slice: [1, 2, 3, 4, 5, 6, 7]
|
||||
slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -246,7 +253,7 @@ jobs:
|
||||
|
||||
# Only this PR's receipts are eligible. No base-branch or cross-PR restore
|
||||
# prefix; every receipt also verifies exact inputs and its original age.
|
||||
- name: Restore this PR's verified judge results
|
||||
- name: Restore this PR's verified judge and E2E results
|
||||
if: github.event_name == 'pull_request'
|
||||
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
|
||||
with:
|
||||
@@ -254,7 +261,7 @@ jobs:
|
||||
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}
|
||||
restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-
|
||||
|
||||
- name: Run slice ${{ matrix.slice }}/7
|
||||
- name: Run slice ${{ matrix.slice }}
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
@@ -285,7 +292,7 @@ jobs:
|
||||
|
||||
# An unrelated failing case does not discard already verified passes.
|
||||
# Failed/retried/partial attempts never become receipts in the first place.
|
||||
- name: Save verified judge results for this PR
|
||||
- name: Save verified judge and E2E results for this PR
|
||||
if: ${{ !cancelled() && steps.receipts.outputs.present == 'true' }}
|
||||
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
|
||||
with:
|
||||
@@ -531,7 +538,7 @@ jobs:
|
||||
</details>
|
||||
|
||||
---
|
||||
*Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → 6 executors → fail-closed report). Reused scores retain their original provenance and expiry.*"
|
||||
*Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → duration-packed executors → fail-closed report). Reused results retain their original provenance and expiry.*"
|
||||
|
||||
if [ "$FAILED" -gt 0 ]; then
|
||||
FAILURES=""
|
||||
|
||||
@@ -1,44 +1,178 @@
|
||||
{
|
||||
"version": 1,
|
||||
"recordedAt": "2026-09-28T23:00:00Z",
|
||||
"durations": {
|
||||
"test/skill-e2e-ask-user-question-format-compliance.test.ts": 67000,
|
||||
"test/skill-e2e-bws.test.ts": 99000,
|
||||
"test/skill-e2e-coverage-audit.test.ts": 60000,
|
||||
"test/skill-e2e-cso.test.ts": 220000,
|
||||
"test/skill-e2e-deploy.test.ts": 496000,
|
||||
"test/skill-e2e-design.test.ts": 228000,
|
||||
"test/skill-e2e-diagram.test.ts": 25000,
|
||||
"test/skill-e2e-docsync-spawned.test.ts": 49000,
|
||||
"test/skill-e2e-hermetic-canary.test.ts": 5000,
|
||||
"test/skill-e2e-investigate-owned-completion.test.ts": 50000,
|
||||
"test/skill-e2e-investigate-owned-termination.test.ts": 49000,
|
||||
"test/skill-e2e-learnings.test.ts": 32000,
|
||||
"test/skill-e2e-office-hours-auto-mode.test.ts": 61000,
|
||||
"test/skill-e2e-plan-ceo-finding-floor.test.ts": 233000,
|
||||
"test/skill-e2e-plan-ceo-plan-mode.test.ts": 35000,
|
||||
"test/skill-e2e-plan-design-with-ui.test.ts": 435000,
|
||||
"test/skill-e2e-plan-devex-finding-floor.test.ts": 187000,
|
||||
"test/skill-e2e-plan-devex-plan-mode.test.ts": 103000,
|
||||
"test/skill-e2e-plan-mode-no-op.test.ts": 206000,
|
||||
"test/skill-e2e-plan-tune.test.ts": 58000,
|
||||
"test/skill-e2e-plan.test.ts": 312000,
|
||||
"test/skill-e2e-qa-workflow.test.ts": 437000,
|
||||
"test/skill-e2e-retro.test.ts": 139000,
|
||||
"test/skill-e2e-review-army.test.ts": 568000,
|
||||
"test/skill-e2e-review-attribution.test.ts": 80000,
|
||||
"test/skill-e2e-review.test.ts": 135000,
|
||||
"test/skill-e2e-session-intelligence.test.ts": 47000,
|
||||
"test/skill-e2e-shared-libs-paths.test.ts": 713000,
|
||||
"test/skill-e2e-shared-libs.test.ts": 662000,
|
||||
"test/skill-e2e-ship-docsync.test.ts": 148000,
|
||||
"test/skill-e2e-ship-hook-consent.test.ts": 72000,
|
||||
"test/skill-e2e-ship-hook-refresh.test.ts": 66000,
|
||||
"test/skill-e2e-skillify.test.ts": 156000,
|
||||
"test/skill-e2e-third-party-actions.test.ts": 75000,
|
||||
"test/skill-e2e-triage.test.ts": 49000,
|
||||
"test/skill-e2e-workflow.test.ts": 234000,
|
||||
"test/skill-llm-eval.test.ts": 103000,
|
||||
"test/skill-routing-e2e.test.ts": 1000
|
||||
"version": 2,
|
||||
"recordedAt": "2026-09-28T09:05:00Z",
|
||||
"source": "periodic census run 36385945043 (eval-slices = periodic, gate-census = gate); timed-out shards record their wall, self-skipping shards 1s; case shards (<file>#<case>) from the Bun per-case walls in that run (gate census log) and the eval JSON slowest cases (periodic)",
|
||||
"tiers": {
|
||||
"gate": {
|
||||
"test/llm-judge-recommendation.test.ts": 1000,
|
||||
"test/skill-e2e-ask-user-question-format-compliance.test.ts": 60000,
|
||||
"test/skill-e2e-autoplan-dual-voice.test.ts": 1000,
|
||||
"test/skill-e2e-bws.test.ts": 82000,
|
||||
"test/skill-e2e-context-skills.test.ts": 1000,
|
||||
"test/skill-e2e-coverage-audit.test.ts": 46000,
|
||||
"test/skill-e2e-cso.test.ts": 251000,
|
||||
"test/skill-e2e-deploy.test.ts": 427000,
|
||||
"test/skill-e2e-design.test.ts": 261000,
|
||||
"test/skill-e2e-design.test.ts#design-review-detector-shim": 51000,
|
||||
"test/skill-e2e-design.test.ts#design-review-detector-shim-dom": 108000,
|
||||
"test/skill-e2e-design.test.ts#design-review-plugin-handoff": 110000,
|
||||
"test/skill-e2e-design.test.ts#plan-design-review-no-ui-scope": 43000,
|
||||
"test/skill-e2e-diagram.test.ts": 32000,
|
||||
"test/skill-e2e-docsync-spawned.test.ts": 51000,
|
||||
"test/skill-e2e-first-task-scaffold.test.ts": 1000,
|
||||
"test/skill-e2e-gbrain-roundtrip-local.test.ts": 1000,
|
||||
"test/skill-e2e-hermetic-canary.test.ts": 8000,
|
||||
"test/skill-e2e-investigate-owned-completion.test.ts": 45000,
|
||||
"test/skill-e2e-investigate-owned-termination.test.ts": 44000,
|
||||
"test/skill-e2e-ios-device.test.ts": 1000,
|
||||
"test/skill-e2e-learnings.test.ts": 35000,
|
||||
"test/skill-e2e-office-hours-auto-mode.test.ts": 67000,
|
||||
"test/skill-e2e-office-hours-brain-writeback.test.ts": 1000,
|
||||
"test/skill-e2e-office-hours-phase4.test.ts": 1000,
|
||||
"test/skill-e2e-office-hours.test.ts": 1000,
|
||||
"test/skill-e2e-plan-ceo-finding-floor.test.ts": 516000,
|
||||
"test/skill-e2e-plan-ceo-plan-mode.test.ts": 39000,
|
||||
"test/skill-e2e-plan-decision-classification.test.ts": 1000,
|
||||
"test/skill-e2e-plan-design-with-ui.test.ts": 500000,
|
||||
"test/skill-e2e-plan-devex-finding-floor.test.ts": 233000,
|
||||
"test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 1000,
|
||||
"test/skill-e2e-plan-devex-plan-mode.test.ts": 85000,
|
||||
"test/skill-e2e-plan-format.test.ts": 1000,
|
||||
"test/skill-e2e-plan-mode-no-op.test.ts": 206000,
|
||||
"test/skill-e2e-plan-prosons.test.ts": 1000,
|
||||
"test/skill-e2e-plan-tune.test.ts": 60000,
|
||||
"test/skill-e2e-plan.test.ts": 251000,
|
||||
"test/skill-e2e-plan.test.ts#codex-offered-ceo-review": 55000,
|
||||
"test/skill-e2e-plan.test.ts#codex-offered-design-review": 59000,
|
||||
"test/skill-e2e-plan.test.ts#codex-offered-eng-review": 58000,
|
||||
"test/skill-e2e-plan.test.ts#codex-offered-office-hours": 54000,
|
||||
"test/skill-e2e-plan.test.ts#office-hours-spec-review": 34000,
|
||||
"test/skill-e2e-plan.test.ts#plan-ceo-review-benefits": 52000,
|
||||
"test/skill-e2e-plan.test.ts#plan-review-report": 51000,
|
||||
"test/skill-e2e-qa-bugs.test.ts": 1000,
|
||||
"test/skill-e2e-qa-workflow.test.ts": 398000,
|
||||
"test/skill-e2e-retro.test.ts": 152000,
|
||||
"test/skill-e2e-review-army.test.ts": 520000,
|
||||
"test/skill-e2e-review-army.test.ts#review-army-delivery-audit": 48000,
|
||||
"test/skill-e2e-review-army.test.ts#review-army-json-findings": 24000,
|
||||
"test/skill-e2e-review-army.test.ts#review-army-migration-safety": 140000,
|
||||
"test/skill-e2e-review-army.test.ts#review-army-perf-n-plus-one": 244000,
|
||||
"test/skill-e2e-review-army.test.ts#review-army-quality-score": 64000,
|
||||
"test/skill-e2e-review-attribution.test.ts": 87000,
|
||||
"test/skill-e2e-review.test.ts": 129000,
|
||||
"test/skill-e2e-session-intelligence.test.ts": 59000,
|
||||
"test/skill-e2e-shared-libs-paths.test.ts": 680000,
|
||||
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-index-flags": 201000,
|
||||
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-path-eligibility": 252000,
|
||||
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-prior-coverage": 227000,
|
||||
"test/skill-e2e-shared-libs.test.ts": 658000,
|
||||
"test/skill-e2e-shared-libs.test.ts#shared-libs-read-only": 230000,
|
||||
"test/skill-e2e-shared-libs.test.ts#shared-libs-review-lifecycle": 278000,
|
||||
"test/skill-e2e-shared-libs.test.ts#shared-libs-review-revalidation": 427000,
|
||||
"test/skill-e2e-shared-libs.test.ts#shared-libs-unsupported-git": 168000,
|
||||
"test/skill-e2e-ship-docsync.test.ts": 129000,
|
||||
"test/skill-e2e-ship-hook-consent.test.ts": 63000,
|
||||
"test/skill-e2e-ship-hook-refresh.test.ts": 70000,
|
||||
"test/skill-e2e-skillify.test.ts": 188000,
|
||||
"test/skill-e2e-third-party-actions.test.ts": 70000,
|
||||
"test/skill-e2e-triage.test.ts": 80000,
|
||||
"test/skill-e2e-workflow.test.ts": 300000,
|
||||
"test/skill-llm-eval.test.ts": 354000,
|
||||
"test/skill-routing-e2e.test.ts": 1000
|
||||
},
|
||||
"periodic": {
|
||||
"test/carve-section-loading-browse.test.ts": 109000,
|
||||
"test/carve-section-loading-codex.test.ts": 118000,
|
||||
"test/carve-section-loading-design-consultation.test.ts": 349000,
|
||||
"test/carve-section-loading-design-html.test.ts": 259000,
|
||||
"test/carve-section-loading-design-shotgun.test.ts": 194000,
|
||||
"test/carve-section-loading-document-release.test.ts": 95000,
|
||||
"test/carve-section-loading-land-and-deploy.test.ts": 185000,
|
||||
"test/carve-section-loading-plan-design-review.test.ts": 284000,
|
||||
"test/carve-section-loading-plan-devex-review.test.ts": 396000,
|
||||
"test/carve-section-loading-plan-eng-review.test.ts": 301000,
|
||||
"test/carve-section-loading-qa.test.ts": 114000,
|
||||
"test/carve-section-loading-retro.test.ts": 230000,
|
||||
"test/carve-section-loading-review.test.ts": 174000,
|
||||
"test/carve-section-loading-setup-gbrain.test.ts": 91000,
|
||||
"test/carve-section-loading-spec.test.ts": 150000,
|
||||
"test/codex-e2e-recommendation-substance.test.ts": 1000,
|
||||
"test/codex-e2e-shared-libs.test.ts": 1000,
|
||||
"test/codex-e2e-sol-scope.test.ts": 1000,
|
||||
"test/codex-e2e.test.ts": 1000,
|
||||
"test/llm-judge-recommendation.test.ts": 15000,
|
||||
"test/skill-e2e-arm-benchmark.test.ts": 76000,
|
||||
"test/skill-e2e-aside.test.ts": 1000,
|
||||
"test/skill-e2e-auq-consistency.test.ts": 75000,
|
||||
"test/skill-e2e-auq-matrix.test.ts": 320000,
|
||||
"test/skill-e2e-auq-verbose-vs-carved-ab.test.ts": 66000,
|
||||
"test/skill-e2e-auto-decide-preserved.test.ts": 113000,
|
||||
"test/skill-e2e-autoplan-dual-voice.test.ts": 532000,
|
||||
"test/skill-e2e-benchmark-providers.test.ts": 9000,
|
||||
"test/skill-e2e-bws.test.ts": 1000,
|
||||
"test/skill-e2e-context-skills.test.ts": 181000,
|
||||
"test/skill-e2e-coverage-audit.test.ts": 1000,
|
||||
"test/skill-e2e-cso.test.ts": 358000,
|
||||
"test/skill-e2e-deploy.test.ts": 1000,
|
||||
"test/skill-e2e-design.test.ts": 817000,
|
||||
"test/skill-e2e-design.test.ts#design-consultation-core": 204000,
|
||||
"test/skill-e2e-design.test.ts#design-html-slop-gate": 205000,
|
||||
"test/skill-e2e-design.test.ts#plan-design-review-plan-mode": 283000,
|
||||
"test/skill-e2e-diagram.test.ts": 166000,
|
||||
"test/skill-e2e-first-task-scaffold.test.ts": 9000,
|
||||
"test/skill-e2e-gbrain-roundtrip-local.test.ts": 1000,
|
||||
"test/skill-e2e-health.test.ts": 165000,
|
||||
"test/skill-e2e-hermetic-canary.test.ts": 1000,
|
||||
"test/skill-e2e-ios-device.test.ts": 1000,
|
||||
"test/skill-e2e-learnings.test.ts": 1000,
|
||||
"test/skill-e2e-office-hours-brain-writeback.test.ts": 174000,
|
||||
"test/skill-e2e-office-hours-phase4.test.ts": 60000,
|
||||
"test/skill-e2e-office-hours-section-loading.test.ts": 1200000,
|
||||
"test/skill-e2e-office-hours.test.ts": 119000,
|
||||
"test/skill-e2e-outside-plan-disabled.test.ts": 43000,
|
||||
"test/skill-e2e-outside-voice.test.ts": 1000,
|
||||
"test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts": 81000,
|
||||
"test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts": 66000,
|
||||
"test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts": 42000,
|
||||
"test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts": 278000,
|
||||
"test/skill-e2e-plan-ceo-mode-routing.test.ts": 575000,
|
||||
"test/skill-e2e-plan-ceo-review-section-loading.test.ts": 604000,
|
||||
"test/skill-e2e-plan-ceo-split-overflow.test.ts": 1332000,
|
||||
"test/skill-e2e-plan-decision-classification.test.ts": 122000,
|
||||
"test/skill-e2e-plan-design-finding-floor.test.ts": 175000,
|
||||
"test/skill-e2e-plan-design-plan-mode.test.ts": 117000,
|
||||
"test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 83000,
|
||||
"test/skill-e2e-plan-eng-finding-floor.test.ts": 282000,
|
||||
"test/skill-e2e-plan-eng-multi-finding-batching.test.ts": 734000,
|
||||
"test/skill-e2e-plan-eng-plan-mode.test.ts": 91000,
|
||||
"test/skill-e2e-plan-format.test.ts": 282000,
|
||||
"test/skill-e2e-plan-prosons.test.ts": 180000,
|
||||
"test/skill-e2e-plan-tune.test.ts": 1000,
|
||||
"test/skill-e2e-plan.test.ts": 824000,
|
||||
"test/skill-e2e-plan.test.ts#plan-ceo-review": 123000,
|
||||
"test/skill-e2e-plan.test.ts#plan-ceo-review-selective": 244000,
|
||||
"test/skill-e2e-plan.test.ts#plan-eng-review-artifact": 153000,
|
||||
"test/skill-e2e-qa-bugs.test.ts": 283000,
|
||||
"test/skill-e2e-qa-workflow.test.ts": 239000,
|
||||
"test/skill-e2e-retro.test.ts": 114000,
|
||||
"test/skill-e2e-review-army.test.ts": 511000,
|
||||
"test/skill-e2e-review-army.test.ts#review-army-consensus": 278000,
|
||||
"test/skill-e2e-review-army.test.ts#review-army-simplification": 144000,
|
||||
"test/skill-e2e-review-attribution.test.ts": 1000,
|
||||
"test/skill-e2e-review.test.ts": 131000,
|
||||
"test/skill-e2e-session-intelligence.test.ts": 1000,
|
||||
"test/skill-e2e-setup-gbrain-bad-token.test.ts": 49000,
|
||||
"test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts": 119000,
|
||||
"test/skill-e2e-setup-gbrain-remote.test.ts": 110000,
|
||||
"test/skill-e2e-shared-libs-periodic.test.ts": 415000,
|
||||
"test/skill-e2e-ship-section-loading.test.ts": 379000,
|
||||
"test/skill-e2e-skillify.test.ts": 56000,
|
||||
"test/skill-e2e-sync-gbrain-readiness.test.ts": 74000,
|
||||
"test/skill-e2e-third-party-actions.test.ts": 1000,
|
||||
"test/skill-e2e-triage.test.ts": 1000,
|
||||
"test/skill-e2e-workflow.test.ts": 1000,
|
||||
"test/skill-llm-eval.test.ts": 402000,
|
||||
"test/skill-routing-e2e.test.ts": 57000
|
||||
}
|
||||
}
|
||||
}
|
||||
+464
-115
@@ -67,12 +67,15 @@ import {
|
||||
} from './test-strict-output';
|
||||
import { PAID_TEST_GLOBS, isPaidTestFile } from '../test/helpers/paid-test-set';
|
||||
import { PERIODIC_CI_EXCLUDE } from '../test/helpers/periodic-exclude-data';
|
||||
import { FILE_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets';
|
||||
import { FILE_RETRY_BUDGETS, SHORT_CASE_RETRY_FILES, STRICT_RETRY_CASE_BUDGETS } from '../test/helpers/eval-budgets';
|
||||
import { getProjectEvalDir, getClaudeCliVersion, isFinalizedEvalResultFile, evalEntryOutcome } from '../test/helpers/eval-store';
|
||||
import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-review';
|
||||
import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight';
|
||||
import { OVERLAY_MIN_FILE_WALL_MS } from '../test/helpers/overlay-case-policy';
|
||||
import { PR_PROFILE_CASE_IDS, PR_PROFILE_FILES, packageChangeOnlyVersion, selectPrProfile, type PrProfileSelection } from './test-pr-profile';
|
||||
import { e2eReuseLaneProblem, prepareE2EShardReuse } from './e2e-shard-reuse';
|
||||
|
||||
type E2EShardReuse = NonNullable<ReturnType<typeof prepareE2EShardReuse>>;
|
||||
import {
|
||||
detectBaseBranch,
|
||||
getChangedFiles,
|
||||
@@ -88,7 +91,8 @@ export { PERIODIC_CI_EXCLUDE };
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
|
||||
export type PaidTier = 'gate' | 'periodic';
|
||||
export type PaidTier = 'gate' | 'periodic' | 'marathon';
|
||||
export const PAID_TIERS: readonly PaidTier[] = ['gate', 'periodic', 'marathon'];
|
||||
export type PaidProfile = 'pr' | 'full';
|
||||
|
||||
export interface PaidCaseSelection {
|
||||
@@ -117,6 +121,64 @@ export function isOverlayTestFile(file: string): boolean {
|
||||
return /^skill-e2e-overlay-harness-.+\.test\.ts$/.test(path.basename(normalizeRelativePath(file)));
|
||||
}
|
||||
|
||||
/**
|
||||
* Files whose cases run in separate processes, one shard per registered E2E
|
||||
* case (`<file>#<case id>`): the file's lane wall exceeds one runner's budget
|
||||
* while every case is short. Separate processes also give each case its own
|
||||
* SDK semaphore, so shared-libs(-paths) capture waves never queue inside a
|
||||
* sibling case's wall (the reason paths runs test.serial in one process).
|
||||
* Every case must be a registered, literal E2E id whose Bun test name is the
|
||||
* id or its CASE_TEST_NAMES label (test/paid-shards.test.ts scans the sources).
|
||||
*/
|
||||
export const CASE_SHARDED_FILES: readonly string[] = [
|
||||
'test/skill-e2e-design.test.ts',
|
||||
'test/skill-e2e-plan.test.ts',
|
||||
'test/skill-e2e-review-army.test.ts',
|
||||
'test/skill-e2e-shared-libs-paths.test.ts',
|
||||
'test/skill-e2e-shared-libs.test.ts',
|
||||
];
|
||||
|
||||
/** Bun test names that differ from their E2E id. */
|
||||
export const CASE_TEST_NAMES: Record<string, string> = {
|
||||
'plan-review-report': '/plan-eng-review writes GSTACK REVIEW REPORT to plan file',
|
||||
'auq-format-gate': "/plan-ceo-review's first AskUserQuestion is a compliant decision brief (7/7 + substance)",
|
||||
};
|
||||
|
||||
const CASE_KEY_SEPARATOR = '#';
|
||||
|
||||
/** The test file behind a shard key (`<file>` or `<file>#<case id>`). */
|
||||
export function shardFile(key: string): string {
|
||||
return normalizeRelativePath(key).split(CASE_KEY_SEPARATOR)[0]!;
|
||||
}
|
||||
|
||||
/** The E2E case id of a case shard key, else null. */
|
||||
export function shardCaseId(key: string): string | null {
|
||||
const [, id] = normalizeRelativePath(key).split(CASE_KEY_SEPARATOR);
|
||||
return id ?? null;
|
||||
}
|
||||
|
||||
/** Exact Bun name pattern for a set of case ids (labels where the test name differs). */
|
||||
export function caseTestNamePattern(ids: string[]): string {
|
||||
const escaped = ids.map(id => (CASE_TEST_NAMES[id] ?? id).replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
|
||||
return `(?:^|\\s)(?:${escaped.join('|')})$`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Replace each case-sharded file with one key per registered case of `tier`.
|
||||
* Throws when such a file's registration is not statically complete: an
|
||||
* unregistered case would otherwise silently never run.
|
||||
*/
|
||||
export function expandCaseShards(files: string[], tier: PaidTier, rootDir = ROOT,
|
||||
touchfiles: Record<string, string[]> = E2E_TOUCHFILES, tiers: Record<string, string> = E2E_TIERS): string[] {
|
||||
return files.flatMap(file => {
|
||||
const rel = normalizeRelativePath(file);
|
||||
if (!CASE_SHARDED_FILES.includes(rel)) return [file];
|
||||
const { registered, known } = fileCaseRegistration(rel, fs.readFileSync(path.join(rootDir, rel), 'utf8'), touchfiles, tiers);
|
||||
if (!known) throw new Error(`Case-sharded ${rel} needs a complete literal case registration`);
|
||||
return registered.filter(id => tiers[id] === tier).sort().map(id => `${rel}${CASE_KEY_SEPARATOR}${id}`);
|
||||
});
|
||||
}
|
||||
|
||||
/** Compatibility helper for callers that only need the effective wall. */
|
||||
export function resolvePaidShardTimeoutMs(files: string[], explicitTimeoutMs?: number): number {
|
||||
return resolvePaidShardBudget(files, explicitTimeoutMs).timeoutMs;
|
||||
@@ -157,21 +219,47 @@ export interface TierClassification {
|
||||
* guard runs and self-skips.
|
||||
*/
|
||||
export function classifyPaidTestFile(source: string, tier: PaidTier): TierClassification {
|
||||
const other: PaidTier = tier === 'gate' ? 'periodic' : 'gate';
|
||||
const declares = (candidate: PaidTier) =>
|
||||
new RegExp(`EVALS_TIER\\s*===\\s*['"\`]${candidate}['"\`]`).test(source) ||
|
||||
new RegExp(`\\b(?:describeE2ETier|e2eTierEnabled)\\(\\s*['"\`]${candidate}['"\`]`).test(source);
|
||||
|
||||
if (declares(tier)) return { included: true, reason: `declares tier '${tier}'` };
|
||||
if (declares(other)) return { included: false, reason: `declares tier '${other}' only` };
|
||||
const others = PAID_TIERS.filter(candidate => candidate !== tier && declares(candidate));
|
||||
if (others.length) return { included: false, reason: `declares tier ${others.map(other => `'${other}'`).join(' and ')} only` };
|
||||
return { included: true, reason: 'no whole-file tier guard — runtime E2E_TIERS filter decides' };
|
||||
}
|
||||
|
||||
/**
|
||||
* The E2E ids a paid file registers: the touchfile registrations that list the
|
||||
* file. `known` is true only when those ids are complete: no computed
|
||||
* registration (testName, *IfSelected, describeIfSelected with a non-literal
|
||||
* argument) and every literal registration argument is among them. Quoted
|
||||
* strings elsewhere (comments, skill paths) never count.
|
||||
*/
|
||||
export function fileCaseRegistration(
|
||||
file: string, source: string,
|
||||
touchfiles: Record<string, string[]> = E2E_TOUCHFILES,
|
||||
tiers: Record<string, string> = E2E_TIERS,
|
||||
): { registered: string[]; known: boolean } {
|
||||
const rel = normalizeRelativePath(file);
|
||||
const registered = Object.keys(touchfiles).filter(key => touchfiles[key]!.includes(rel));
|
||||
const computed = /testName\s*:\s*(?!string\b)(?:`[^`]*\$\{|[A-Za-z_$])/.test(source)
|
||||
|| /\btest(?:Concurrent)?IfSelected\s*\(\s*(?:`[^`]*\$\{|[A-Za-z_$])/.test(source)
|
||||
|| /\bdescribeIfSelected\s*\([^,]*,(?!\s*\[)/.test(source)
|
||||
|| [...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)].some(m => m[1]!.split(',')
|
||||
.map(item => item.trim()).some(item => item && !/^(['"`])[^'"`$]*\1$/.test(item)));
|
||||
const literal = [
|
||||
...[...source.matchAll(/testName\s*:\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!),
|
||||
...[...source.matchAll(/\btest(?:Concurrent)?IfSelected\s*\(\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!),
|
||||
...[...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)]
|
||||
.flatMap(m => [...m[1]!.matchAll(/(['"`])([^'"`]+)\1/g)].map(n => n[2]!)),
|
||||
].filter(id => id in tiers);
|
||||
return { registered, known: registered.length > 0 && !computed && literal.every(id => registered.includes(id)) };
|
||||
}
|
||||
|
||||
/**
|
||||
* A file is skipped for a tier lane only when its registered E2E ids are fully
|
||||
* known and none of them has that tier. Ids are the touchfile registrations that
|
||||
* list the file plus literal registration arguments (testName, *IfSelected);
|
||||
* quoted strings elsewhere (comments, skill paths) never count. Any computed
|
||||
* known (fileCaseRegistration) and none of them has that tier. Any computed
|
||||
* registration, an id missing from the file's touchfile registration, or no id at
|
||||
* all keeps today's scheduling (the child's runtime filter decides).
|
||||
*/
|
||||
@@ -180,26 +268,27 @@ export function tierSkipReason(
|
||||
touchfiles: Record<string, string[]> = E2E_TOUCHFILES,
|
||||
tiers: Record<string, string> = E2E_TIERS,
|
||||
): string | null {
|
||||
const rel = normalizeRelativePath(file);
|
||||
const registered = Object.keys(touchfiles).filter(key => touchfiles[key]!.includes(rel));
|
||||
if (!registered.length) return null;
|
||||
const computed = /testName\s*:\s*(?:`[^`]*\$\{|[A-Za-z_$])/.test(source)
|
||||
|| /\btest(?:Concurrent)?IfSelected\s*\(\s*(?:`[^`]*\$\{|[A-Za-z_$])/.test(source)
|
||||
|| /\bdescribeIfSelected\s*\([^,]*,(?!\s*\[)/.test(source)
|
||||
|| [...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)].some(m => m[1]!.split(',')
|
||||
.map(item => item.trim()).some(item => item && !/^(['"`])[^'"`$]*\1$/.test(item)));
|
||||
if (computed) return null;
|
||||
const literal = [
|
||||
...[...source.matchAll(/testName\s*:\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!),
|
||||
...[...source.matchAll(/\btest(?:Concurrent)?IfSelected\s*\(\s*(['"`])([^'"`]+)\1/g)].map(m => m[2]!),
|
||||
...[...source.matchAll(/\bdescribeIfSelected\s*\([^,]*,\s*\[([^\]]*)\]/g)]
|
||||
.flatMap(m => [...m[1]!.matchAll(/(['"`])([^'"`]+)\1/g)].map(n => n[2]!)),
|
||||
].filter(id => id in tiers);
|
||||
if (literal.some(id => !registered.includes(id))) return null;
|
||||
if (registered.some(id => tiers[id] === tier)) return null;
|
||||
const { registered, known } = fileCaseRegistration(file, source, touchfiles, tiers);
|
||||
if (!known || registered.some(id => tiers[id] === tier)) return null;
|
||||
return `skipped: no E2E_TIERS id has tier ${tier}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* The marathon lane selects positively: a file runs there only when it
|
||||
* declares the marathon tier or registers a marathon-tier case. Files without
|
||||
* marathon work never cost a marathon runner, and gate/periodic files never
|
||||
* gain a third execution.
|
||||
*/
|
||||
export function marathonSkipReason(
|
||||
file: string, source: string,
|
||||
touchfiles: Record<string, string[]> = E2E_TOUCHFILES,
|
||||
tiers: Record<string, string> = E2E_TIERS,
|
||||
): string | null {
|
||||
if (classifyPaidTestFile(source, 'marathon').reason === "declares tier 'marathon'") return null;
|
||||
const { registered } = fileCaseRegistration(file, source, touchfiles, tiers);
|
||||
return registered.some(id => tiers[id] === 'marathon') ? null : 'skipped: declares no marathon tier and registers no marathon case';
|
||||
}
|
||||
|
||||
export interface TierSelection {
|
||||
selected: string[];
|
||||
excluded: Array<{ file: string; reason: string }>;
|
||||
@@ -213,11 +302,11 @@ export function selectPaidTestFiles(files: string[], tier: PaidTier, rootDir = R
|
||||
if (carveSkill && files.some(file => carveWrapper(file)) && !files.some(file => carveWrapper(file) === carveSkill)) {
|
||||
throw new Error(`GSTACK_CARVE_SKILL=${carveSkill} has no generic section-loading wrapper`);
|
||||
}
|
||||
// Periodic-lane exclusions (documented-red / manual-hardware files): a
|
||||
// Scheduled-lane exclusions (documented-red / manual-hardware files): a
|
||||
// known-red weekly shard is triage waste locally AND in CI, so the list
|
||||
// applies to every periodic run, with the reason surfaced per file.
|
||||
// applies to every periodic and marathon run, with the reason surfaced per file.
|
||||
const ciExcluded = (file: string): { reason: string; tracking: string } | undefined =>
|
||||
tier === 'periodic' ? PERIODIC_CI_EXCLUDE[normalizeRelativePath(file)] : undefined;
|
||||
tier !== 'gate' ? PERIODIC_CI_EXCLUDE[normalizeRelativePath(file)] : undefined;
|
||||
for (const file of files) {
|
||||
// One wrapper per process means a child-side return now creates an empty
|
||||
// shard. Apply the existing explicit cost scope before planning processes.
|
||||
@@ -233,7 +322,8 @@ export function selectPaidTestFiles(files: string[], tier: PaidTier, rootDir = R
|
||||
}
|
||||
const source = fs.readFileSync(path.join(rootDir, file), 'utf8');
|
||||
const classification = classifyPaidTestFile(source, tier);
|
||||
const skip = classification.included ? tierSkipReason(file, source, tier) : null;
|
||||
const skip = !classification.included ? null
|
||||
: tier === 'marathon' ? marathonSkipReason(file, source) : tierSkipReason(file, source, tier);
|
||||
if (classification.included && !skip) selected.push(file);
|
||||
else excluded.push({ file, reason: skip ?? classification.reason });
|
||||
}
|
||||
@@ -378,28 +468,29 @@ function packageVersionOnlySinceBase(rootDir: string, baseRef: string): boolean
|
||||
}
|
||||
|
||||
/** Only audited per-case files, plus the separately selected judge, enter the fast profile. */
|
||||
/** The selected PR-profile case ids a shard key owns (a case key owns at most its own case). */
|
||||
function prProfileShardIds(key: string, selection: PaidCaseSelection): string[] {
|
||||
const caseId = shardCaseId(key);
|
||||
return (PR_PROFILE_FILES[shardFile(key)] ?? [])
|
||||
.filter(id => (caseId === null || id === caseId) && (selection.e2e === null || selection.e2e.includes(id)));
|
||||
}
|
||||
|
||||
export function prProfileFileSelected(file: string, selection: PaidCaseSelection): boolean {
|
||||
if (file === 'test/skill-llm-eval.test.ts') return selection.judges === null || selection.judges.length > 0;
|
||||
const ids = PR_PROFILE_FILES[normalizeRelativePath(file)];
|
||||
return !!ids && (selection.e2e === null || ids.some(id => selection.e2e!.includes(id)));
|
||||
return prProfileShardIds(file, selection).length > 0;
|
||||
}
|
||||
|
||||
export function expectedPrCaseCount(file: string, selection: PaidCaseSelection): number {
|
||||
if (file === 'test/skill-llm-eval.test.ts') return selection.judges?.length ?? Object.keys(LLM_JUDGE_TOUCHFILES).length;
|
||||
return (PR_PROFILE_FILES[normalizeRelativePath(file)] ?? []).filter(id => selection.e2e === null || selection.e2e.includes(id)).length;
|
||||
return prProfileShardIds(file, selection).length;
|
||||
}
|
||||
|
||||
export function prProfileTestNamePattern(file: string, selection: PaidCaseSelection): string {
|
||||
const labels: Record<string, string> = {
|
||||
'plan-review-report': '/plan-eng-review writes GSTACK REVIEW REPORT to plan file',
|
||||
'auq-format-gate': "/plan-ceo-review's first AskUserQuestion is a compliant decision brief (7/7 + substance)",
|
||||
};
|
||||
const ids = file === 'test/skill-llm-eval.test.ts'
|
||||
? selection.judges ?? Object.keys(LLM_JUDGE_TOUCHFILES)
|
||||
: (PR_PROFILE_FILES[file] ?? []).filter(id => selection.e2e === null || selection.e2e.includes(id));
|
||||
: prProfileShardIds(file, selection);
|
||||
if (ids.length === 0) throw new Error(`No selected PR cases for ${file}`);
|
||||
const escaped = ids.map(id => (labels[id] ?? id).replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
|
||||
return `(?:^|\\s)(?:${escaped.join('|')})$`;
|
||||
return caseTestNamePattern(ids);
|
||||
}
|
||||
|
||||
export function paidSelectionEnv(profile: PaidProfile, selection: PaidCaseSelection, reason: string): NodeJS.ProcessEnv {
|
||||
@@ -443,6 +534,10 @@ export function diffSkipDecisionForFile(
|
||||
options: DiffSkipOptions = {},
|
||||
): ShardSkipDecision {
|
||||
if (selectedNames === null) return { file, kept: true, reason: 'run-all selection' };
|
||||
const caseId = shardCaseId(file);
|
||||
if (caseId !== null) {
|
||||
return selectedNames.has(caseId) ? { file, kept: true, reason: `selected: ${caseId}` } : { file, kept: false, reason: `case ${caseId} not selected` };
|
||||
}
|
||||
const rel = normalizeRelativePath(file);
|
||||
if (!/^test\/skill-e2e-.*\.test\.ts$/.test(rel)) {
|
||||
return { file, kept: true, reason: 'non-skill-e2e paid file — child self-skip authoritative' };
|
||||
@@ -503,7 +598,7 @@ export function planPaidShards(
|
||||
const shards: string[][] = [];
|
||||
let pending: string[] = [];
|
||||
for (const file of unique) {
|
||||
if (isOverlayTestFile(file) || FILE_RETRY_BUDGETS.some(budget => budget.file === file)) {
|
||||
if (isOverlayTestFile(file) || shardCaseId(file) !== null || FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file))) {
|
||||
if (pending.length) shards.push(pending);
|
||||
pending = [];
|
||||
shards.push([file]);
|
||||
@@ -524,7 +619,7 @@ export interface PaidShardBudget {
|
||||
|
||||
/** Explicit caller limits win; registered supervision preserves existing attempts. */
|
||||
export function resolvePaidShardBudget(files: string[], overrideMs?: number): PaidShardBudget {
|
||||
const finding = FILE_RETRY_BUDGETS.find(budget => files.map(normalizeRelativePath).includes(budget.file));
|
||||
const finding = FILE_RETRY_BUDGETS.find(budget => files.map(shardFile).includes(budget.file));
|
||||
if (finding && files.length !== 1) throw new Error('Registered retry budget requires its own shard');
|
||||
if (overrideMs !== undefined && (!Number.isSafeInteger(overrideMs) || overrideMs <= 0 || overrideMs > 2_147_483_647)) {
|
||||
throw new Error('Shard timeout must be a finite positive timer-safe integer');
|
||||
@@ -534,8 +629,11 @@ export function resolvePaidShardBudget(files: string[], overrideMs?: number): Pa
|
||||
if (overlay && overrideMs !== undefined && overrideMs < OVERLAY_MIN_FILE_WALL_MS) {
|
||||
throw new Error(`Overlay shard requires at least ${OVERLAY_MIN_FILE_WALL_MS}ms; explicit wall ${overrideMs}ms cannot preserve its work and finalization budget`);
|
||||
}
|
||||
// A registered file's case shard supervises one case and its allowed attempts.
|
||||
const registeredMs = finding && shardCaseId(files[0]!) !== null
|
||||
? finding.caseMs * (finding.retries + 1) + finding.shardReserveMs : finding?.shardMs;
|
||||
return {
|
||||
timeoutMs: overrideMs ?? (finding ? finding.shardMs : overlay ? OVERLAY_MIN_FILE_WALL_MS : DEFAULT_SHARD_TIMEOUT_MS),
|
||||
timeoutMs: overrideMs ?? (registeredMs ?? (overlay ? OVERLAY_MIN_FILE_WALL_MS : DEFAULT_SHARD_TIMEOUT_MS)),
|
||||
source: overrideMs !== undefined ? 'explicit' : finding ? 'registered' : 'default',
|
||||
policyId: finding?.id ?? null,
|
||||
};
|
||||
@@ -554,8 +652,8 @@ export function buildPaidShardArgs(
|
||||
// Explicit --concurrent/--max-concurrency: the legacy path always set one;
|
||||
// omitting it here made within-shard parallelism differ silently between
|
||||
// the two runners (observed: 1.6x sumdur/wall sharded vs 8x legacy).
|
||||
// Retries default to 1; RETRY_OVERRIDES membership (old matrix rows'
|
||||
// earned `retries: 2`) flows through retriesForFiles at the call site.
|
||||
// Retries come from retriesForFiles (the timeout-is-a-verdict rule) at the
|
||||
// call site; the fallback of 1 serves only direct callers.
|
||||
return ['test', ...files, '--retry', String(retries ?? 1), '--concurrent', `--max-concurrency=${maxConcurrency}`, `--timeout=${timeoutMs}`];
|
||||
}
|
||||
|
||||
@@ -565,7 +663,8 @@ export function buildPaidShardArgs(
|
||||
*/
|
||||
export function shardSlug(files: string[]): string {
|
||||
return files
|
||||
.map((file) => path.basename(normalizeRelativePath(file)).replace(/\.test\.(?:[cm]?[jt]s|tsx|jsx)$/, ''))
|
||||
.map((file) => path.basename(shardFile(file)).replace(/\.test\.(?:[cm]?[jt]s|tsx|jsx)$/, '')
|
||||
+ (shardCaseId(file) === null ? '' : `--${shardCaseId(file)}`))
|
||||
.join('+')
|
||||
.replace(/[^a-zA-Z0-9._+-]/g, '-');
|
||||
}
|
||||
@@ -599,6 +698,8 @@ export interface ShardOutcome {
|
||||
skippedTests: number | null;
|
||||
/** Effective supervised wall; absent only for unstarted or legacy outcomes. */
|
||||
budget?: PaidShardBudget;
|
||||
/** Present when a verified receipt replaced execution (PR lane only). */
|
||||
reused?: { inputKey: string; runId: string; revision: string; completedAt: number };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -655,6 +756,10 @@ export interface RunShardsOptions {
|
||||
/** Fast-profile census: selected real cases per file, excluding Bun skips. */
|
||||
expectedCases?: Record<string, number>;
|
||||
casePatterns?: Record<string, string>;
|
||||
/** The selected case ids per shard key (reported for reused shards). */
|
||||
expectedCaseIds?: Record<string, string[]>;
|
||||
/** PR lane only: verified reuse for one shard's exact child environment and wall. */
|
||||
reuseFor?: (files: string[], env: NodeJS.ProcessEnv, budget: PaidShardBudget) => E2EShardReuse | null;
|
||||
}
|
||||
|
||||
let shardLogSequence = 0;
|
||||
@@ -703,17 +808,10 @@ export async function runPaidShard(
|
||||
const log = options.log ?? ((line: string) => console.log(line));
|
||||
const label = `[test:paid] shard ${shardNumber}/${totalShards}`;
|
||||
|
||||
const { command, args } = options.commandFor
|
||||
? options.commandFor(files)
|
||||
: {
|
||||
command: process.execPath,
|
||||
args: [...buildPaidShardArgs(
|
||||
exactTestFileSelectors(files, rootDir),
|
||||
timeoutMs,
|
||||
options.withinShardConcurrency ?? DEFAULT_WITHIN_SHARD_CONCURRENCY,
|
||||
retriesForFiles(files),
|
||||
), ...(options.casePatterns ? ['--test-name-pattern', options.casePatterns[files[0]]] : [])],
|
||||
};
|
||||
// A case shard runs exactly its one case; PR patterns narrow further.
|
||||
const caseId = files.length === 1 ? shardCaseId(files[0]!) : null;
|
||||
const casePattern = options.casePatterns?.[files[0]!] ?? (caseId !== null ? caseTestNamePattern([caseId]) : undefined);
|
||||
const expectedCases = options.expectedCases ?? (caseId !== null ? { [files[0]!]: 1 } : undefined);
|
||||
|
||||
const env = { ...(options.env ?? process.env) };
|
||||
if (options.evalDirBase) {
|
||||
@@ -725,6 +823,42 @@ export async function runPaidShard(
|
||||
if (!env.GSTACK_CLAUDE_CLI_VERSION) {
|
||||
env.GSTACK_CLAUDE_CLI_VERSION = getClaudeCliVersion();
|
||||
}
|
||||
// Verified first-attempt reuse (PR lane only; scripts/e2e-shard-reuse.ts):
|
||||
// identical consumed inputs to a fresh pass in this PR replace execution
|
||||
// with an explicitly reported reused result.
|
||||
// Bootstrap-retention qualification binds per-run state, so that shard stays fresh.
|
||||
const reuse = files.some(file => normalizeRelativePath(file) === 'test/skill-e2e-qa-workflow.test.ts')
|
||||
? null : options.reuseFor?.(files, env, budget) ?? null;
|
||||
const reused = reuse?.lookup() ?? null;
|
||||
if (reused) {
|
||||
const reusedFrom = { input_key: reused.key, run_id: reused.source.runId, revision: reused.source.revision,
|
||||
completed_at: new Date(reused.source.completedAt).toISOString() };
|
||||
const caseIds = options.expectedCaseIds?.[files[0]!] ?? [];
|
||||
if (env.GSTACK_EVAL_DIR) {
|
||||
fs.mkdirSync(env.GSTACK_EVAL_DIR, { recursive: true });
|
||||
fs.writeFileSync(path.join(env.GSTACK_EVAL_DIR, `e2e-reused-${shardSlug(files)}.json`), `${JSON.stringify({
|
||||
schema_version: 1, tier: 'e2e', shard: shardSlug(files), total_tests: caseIds.length, executed_tests: 0,
|
||||
reused_tests: caseIds.length, passed: caseIds.length, failed: 0, total_cost_usd: 0, total_duration_ms: 0,
|
||||
tests: caseIds.map(name => ({ name, suite: shardSlug(files), tier: 'e2e', passed: true, duration_ms: 0, cost_usd: 0,
|
||||
execution: 'reused', reused_from: reusedFrom, attempt: 1 })),
|
||||
}, null, 2)}\n`);
|
||||
}
|
||||
log(`${label} REUSED ${files.join(' ')} — identical inputs passed in run ${reused.source.runId} at ${reusedFrom.completed_at}`);
|
||||
return { shard: shardNumber, files, status: 'passed', exitCode: 0, elapsedMs: 0, groupPid: null,
|
||||
executedTests: caseIds.length, skippedTests: 0, budget,
|
||||
reused: { inputKey: reused.key, runId: reused.source.runId, revision: reused.source.revision, completedAt: reused.source.completedAt } };
|
||||
}
|
||||
const { command, args } = options.commandFor
|
||||
? options.commandFor(files)
|
||||
: {
|
||||
command: process.execPath,
|
||||
args: [...buildPaidShardArgs(
|
||||
exactTestFileSelectors(files.map(shardFile), rootDir),
|
||||
timeoutMs,
|
||||
options.withinShardConcurrency ?? DEFAULT_WITHIN_SHARD_CONCURRENCY,
|
||||
retriesForFiles(files),
|
||||
), ...(casePattern !== undefined ? ['--test-name-pattern', casePattern] : [])],
|
||||
};
|
||||
// Per-shard temp + Chromium-profile isolation — the free runner treats
|
||||
// this as mandatory (test-free-shards.ts: two concurrent shards on one
|
||||
// profile dir kill each other's browser; shared tmp cross-contaminates),
|
||||
@@ -881,15 +1015,16 @@ export async function runPaidShard(
|
||||
let status: ShardStatus = timedOut
|
||||
? 'timed-out'
|
||||
: !retentionFailed && !logWriteFailed && !incompleteCapture && strictTestExitCode(exitCode ?? 1, summary, expectedFiles) === 0 ? 'passed' : 'failed';
|
||||
if (status === 'passed' && options.expectedCases) {
|
||||
const expected = files.reduce((count, file) => count + (options.expectedCases![file] ?? 0), 0);
|
||||
if (status === 'passed' && expectedCases) {
|
||||
const expected = files.reduce((count, file) => count + (expectedCases[file] ?? 0), 0);
|
||||
const actual = summary.terminalTestCounts.reduce((count, value) => count + value, 0) - summary.skippedTests;
|
||||
if (expected < 1 || actual !== expected) {
|
||||
status = 'failed';
|
||||
log(`${label} expected ${expected} selected cases, executed ${actual}; refusing incomplete PR coverage`);
|
||||
log(`${label} expected ${expected} selected cases, executed ${actual}; refusing incomplete case coverage`);
|
||||
}
|
||||
}
|
||||
const elapsedMs = Date.now() - startedAt;
|
||||
if (status === 'passed' && reuse) reuse.publish();
|
||||
|
||||
// Failure debuggability without the RAM cost: read back only the log's
|
||||
// tail. Live mode already streamed everything, so no re-print there.
|
||||
@@ -1077,6 +1212,16 @@ export interface ManifestEntry {
|
||||
reason?: string;
|
||||
/** Required when a registered retry-budget file is planned. */
|
||||
budget?: PaidShardBudget;
|
||||
/** Budget-mode packing weight (recorded wall, or the whole budget when unknown). */
|
||||
estimatedMs?: number;
|
||||
}
|
||||
|
||||
/** Budget-mode plan: per-executor estimate and the CI job timeout it needs. */
|
||||
export interface PaidSlicePlan {
|
||||
sliceBudgetMs: number;
|
||||
jobs: number;
|
||||
estimatedSliceMs: number[];
|
||||
ciTimeoutMinutes: number;
|
||||
}
|
||||
|
||||
export interface PaidRunManifest {
|
||||
@@ -1089,43 +1234,57 @@ export interface PaidRunManifest {
|
||||
profile?: PaidProfile;
|
||||
selection?: PaidCaseSelection;
|
||||
prCoverage?: PrProfileSelection;
|
||||
plan?: PaidSlicePlan;
|
||||
entries: ManifestEntry[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Files whose old evals.yml matrix rows carried `retries: 2`, with the
|
||||
* receipts that earned them (see the deleted rows' comments). The runner
|
||||
* default stays --retry 1; membership here is a literals map so retry
|
||||
* parity with the matrix is explicit, not folklore.
|
||||
* Automatic retries follow the approved rule in test/helpers/eval-budgets.ts:
|
||||
* a timed-out attempt is a verdict, so only files whose every case budget is
|
||||
* at most RETRY_MAX_CASE_MS keep a retry (registered rows derive it from their
|
||||
* caseMs; SHORT_CASE_RETRY_FILES lists the rest). Overlays and every other
|
||||
* paid file run once. A multi-file shard takes the smallest allowance.
|
||||
*/
|
||||
export const RETRY_OVERRIDES: Record<string, number> = {
|
||||
'test/skill-e2e-workflow.test.ts': 2,
|
||||
'test/skill-e2e-office-hours-auto-mode.test.ts': 2,
|
||||
'test/skill-e2e-plan-mode-no-op.test.ts': 2,
|
||||
};
|
||||
|
||||
export function retriesForFiles(files: string[]): number {
|
||||
if (files.some(isOverlayTestFile)) return 0;
|
||||
return Math.max(1, ...files.map((f) => RETRY_OVERRIDES[normalizeRelativePath(f)] ?? 1));
|
||||
return Math.min(...files.map((file) => {
|
||||
const rel = shardFile(file);
|
||||
const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === rel);
|
||||
if (registered) return registered.retries;
|
||||
return SHORT_CASE_RETRY_FILES.includes(rel) ? 1 : 0;
|
||||
}));
|
||||
}
|
||||
|
||||
export const PAID_TEST_DURATIONS_FILE = 'scripts/paid-test-durations.json';
|
||||
|
||||
/**
|
||||
* Recorded per-file paid-shard wall times (ms) from real CI slice reports,
|
||||
* refreshed with `--report <dir> --write-durations`. A packing hint only: a
|
||||
* missing or corrupt seed keeps the supervision-budget allocation.
|
||||
* Recorded per-file paid-shard wall times (ms) per tier from real CI slice
|
||||
* reports (a file's gate and periodic cases differ), refreshed with
|
||||
* `--report <dir> --write-durations`. A packing hint only: a missing or
|
||||
* corrupt seed keeps the supervision-budget allocation.
|
||||
*/
|
||||
export function loadPaidTestDurations(rootDir = ROOT): Record<string, number> {
|
||||
export function loadPaidTestDurations(rootDir = ROOT, tier: PaidTier = 'gate'): Record<string, number> {
|
||||
try {
|
||||
const parsed = JSON.parse(fs.readFileSync(path.join(rootDir, PAID_TEST_DURATIONS_FILE), 'utf8')) as { durations?: Record<string, unknown> };
|
||||
return Object.fromEntries(Object.entries(parsed.durations ?? {})
|
||||
const parsed = JSON.parse(fs.readFileSync(path.join(rootDir, PAID_TEST_DURATIONS_FILE), 'utf8')) as { version?: unknown; tiers?: Record<string, Record<string, unknown>> };
|
||||
if (parsed.version !== 2) return {};
|
||||
return Object.fromEntries(Object.entries(parsed.tiers?.[tier] ?? {})
|
||||
.filter((entry): entry is [string, number] => typeof entry[1] === 'number' && Number.isFinite(entry[1]) && entry[1] > 0));
|
||||
} catch {
|
||||
return {};
|
||||
}
|
||||
}
|
||||
|
||||
/** Rewrite one tier of the committed seed, keeping the other tiers. */
|
||||
export function writePaidTestDurations(tier: PaidTier, durations: Record<string, number>, rootDir = ROOT): void {
|
||||
const target = path.join(rootDir, PAID_TEST_DURATIONS_FILE);
|
||||
const tiers = Object.fromEntries(PAID_TIERS.map(name => [name, loadPaidTestDurations(rootDir, name)])
|
||||
.filter(([name, recorded]) => name === tier || Object.keys(recorded as object).length > 0));
|
||||
tiers[tier] = durations;
|
||||
const temporary = `${target}.tmp-${process.pid}`;
|
||||
fs.writeFileSync(temporary, `${JSON.stringify({ version: 2, recordedAt: new Date().toISOString(), tiers }, null, 2)}\n`);
|
||||
fs.renameSync(temporary, target);
|
||||
}
|
||||
|
||||
/** Merge a report's executed single-file outcomes into the seed; all-skipped shards carry no cost signal. */
|
||||
export function mergePaidTestDurations(seed: Record<string, number>, results: SliceResult[]): Record<string, number> {
|
||||
const merged = { ...seed };
|
||||
@@ -1138,6 +1297,65 @@ export function mergePaidTestDurations(seed: Record<string, number>, results: Sl
|
||||
return Object.fromEntries(Object.entries(merged).sort(([a], [b]) => (a < b ? -1 : 1)));
|
||||
}
|
||||
|
||||
/** Setup, image pull and artifact upload allowance on top of a slice's supervised wall. */
|
||||
export const CI_SETUP_ALLOWANCE_MINUTES = 20;
|
||||
|
||||
/** Estimated wall of one executor running `files` in order on `jobs` FIFO workers. */
|
||||
export function estimatedSliceMs(files: string[], weight: (file: string) => number, jobs: number): number {
|
||||
const workers = Array<number>(Math.max(1, jobs)).fill(0);
|
||||
for (const file of files) {
|
||||
const next = workers.indexOf(Math.min(...workers));
|
||||
workers[next] += weight(file);
|
||||
}
|
||||
return Math.max(...workers);
|
||||
}
|
||||
|
||||
/** Executor order within a slice: longest recorded work first, then by path. */
|
||||
export function sliceExecutionOrder<T extends { file: string; estimatedMs?: number }>(entries: T[]): T[] {
|
||||
return [...entries].sort((a, b) => (b.estimatedMs ?? 0) - (a.estimatedMs ?? 0) || (a.file < b.file ? -1 : a.file > b.file ? 1 : 0));
|
||||
}
|
||||
|
||||
/** Supervised worst case of one slice in execution order, overlays at their own admission limit. */
|
||||
export function sliceSupervisedWallMs(files: string[], jobs: number, overrideMs?: number): number {
|
||||
return paidShardWallUpperBoundMs(files.filter(file => !isOverlayTestFile(file)), jobs, overrideMs)
|
||||
+ paidShardWallUpperBoundMs(files.filter(isOverlayTestFile), Math.min(jobs, OVERLAY_MAX_ACTIVE_SHARDS), overrideMs);
|
||||
}
|
||||
|
||||
/**
|
||||
* Budget packing: one runner per file or per tightly packed group. Files go
|
||||
* longest-recorded-first into the fullest slice whose estimated wall stays
|
||||
* within the budget (best fit), else into a new slice. A file with no recorded
|
||||
* wall weighs the whole budget, so unknown cost gets a runner of its own. A
|
||||
* file longer than the budget runs alone. Overlay wrappers keep one shared
|
||||
* final slice (one wrapper at a time). The CI timeout covers every slice's
|
||||
* supervised worst case plus the setup allowance.
|
||||
*/
|
||||
export function packBySliceBudget(files: string[], budgetMs: number, jobs: number,
|
||||
recorded: Record<string, number>, timeoutMs?: number): {
|
||||
slices: string[][]; estimates: Record<string, number>; estimatedSliceMs: number[]; ciTimeoutMinutes: number;
|
||||
} {
|
||||
const estimates = Object.fromEntries(files.map(file => [file, recorded[normalizeRelativePath(file)] ?? budgetMs]));
|
||||
const weight = (file: string) => estimates[file]!;
|
||||
const slices: string[][] = [];
|
||||
for (const file of sliceExecutionOrder(files.filter(file => !isOverlayTestFile(file)).map(file => ({ file, estimatedMs: weight(file) }))).map(entry => entry.file)) {
|
||||
let best = -1, bestMs = -1;
|
||||
slices.forEach((planned, index) => {
|
||||
const ms = estimatedSliceMs([...planned, file], weight, jobs);
|
||||
if (ms <= budgetMs && ms > bestMs) { best = index; bestMs = ms; }
|
||||
});
|
||||
if (best < 0) slices.push([file]);
|
||||
else slices[best]!.push(file);
|
||||
}
|
||||
const overlays = files.filter(isOverlayTestFile).sort();
|
||||
if (overlays.length) slices.push(sliceExecutionOrder(overlays.map(file => ({ file, estimatedMs: weight(file) }))).map(entry => entry.file));
|
||||
if (!slices.length) slices.push([]);
|
||||
const estimatedSliceMsList = slices.map(planned => planned.some(isOverlayTestFile)
|
||||
? estimatedSliceMs(planned, weight, Math.min(jobs, OVERLAY_MAX_ACTIVE_SHARDS)) : estimatedSliceMs(planned, weight, jobs));
|
||||
const worst = Math.max(0, ...slices.map(planned => sliceSupervisedWallMs(planned, jobs, timeoutMs)));
|
||||
return { slices, estimates, estimatedSliceMs: estimatedSliceMsList,
|
||||
ciTimeoutMinutes: Math.ceil(worst / 60_000) + CI_SETUP_ALLOWANCE_MINUTES };
|
||||
}
|
||||
|
||||
/** Worker counts whose worst-case slice wall duration packing may never worsen. */
|
||||
export const SUPERVISED_WORKER_COUNTS = [1, 2, 3, 4] as const;
|
||||
|
||||
@@ -1153,7 +1371,13 @@ export const SUPERVISED_WORKER_COUNTS = [1, 2, 3, 4] as const;
|
||||
export function buildRunManifest(opts: {
|
||||
tier: PaidTier;
|
||||
profile?: PaidProfile;
|
||||
sliceCount: number;
|
||||
/** Fixed slice count; exclusive with sliceBudgetMs. */
|
||||
sliceCount?: number;
|
||||
/** Budget mode: pack recorded work so each executor's estimated wall stays
|
||||
* within this budget; the slice count follows from the plan. */
|
||||
sliceBudgetMs?: number;
|
||||
/** Shard workers per executor (EVALS_JOBS) that budget mode plans for. */
|
||||
jobs?: number;
|
||||
evalsAll: boolean;
|
||||
timeoutMs?: number;
|
||||
discovered?: string[];
|
||||
@@ -1165,7 +1389,12 @@ export function buildRunManifest(opts: {
|
||||
/** Weekly gate census only: LLM judges already run in the periodic census and PR gate lanes. */
|
||||
skipJudges?: boolean;
|
||||
}): PaidRunManifest {
|
||||
if (!Number.isInteger(opts.sliceCount) || opts.sliceCount <= 0) {
|
||||
const budgetMode = opts.sliceBudgetMs !== undefined;
|
||||
if (budgetMode === (opts.sliceCount !== undefined)) throw new Error('Plan with exactly one of --slices or --slice-budget');
|
||||
if (budgetMode && (!Number.isSafeInteger(opts.sliceBudgetMs) || opts.sliceBudgetMs! <= 0 || !Number.isSafeInteger(opts.jobs) || opts.jobs! <= 0)) {
|
||||
throw new Error('--slice-budget needs a positive budget and an explicit positive --jobs');
|
||||
}
|
||||
if (!budgetMode && (!Number.isInteger(opts.sliceCount) || opts.sliceCount! <= 0)) {
|
||||
throw new Error(`--slices needs a positive integer. Received: ${opts.sliceCount}`);
|
||||
}
|
||||
const rootDir = opts.rootDir ?? ROOT;
|
||||
@@ -1178,7 +1407,7 @@ export function buildRunManifest(opts: {
|
||||
const selected = opts.skipJudges ? tierSelection.selected.filter(file => !judge(file)) : tierSelection.selected;
|
||||
const excluded = [...tierSelection.excluded, ...(opts.skipJudges ? tierSelection.selected.filter(judge)
|
||||
.map(file => ({ file, reason: 'skipped: LLM judges run in the periodic census and PR gate lanes' })) : [])];
|
||||
const shards = planPaidShards(selected, { maxFilesPerShard: 1 });
|
||||
const shards = planPaidShards(expandCaseShards(selected, opts.tier, rootDir), { maxFilesPerShard: 1 });
|
||||
const cases = computePaidCaseSelection({ profile, env, rootDir, changedFiles: opts.changedFiles });
|
||||
const fast = cases.coverage?.mode === 'pr';
|
||||
const profileShards = fast ? shards.filter(files => prProfileFileSelected(files[0], cases.selection)) : shards;
|
||||
@@ -1189,14 +1418,32 @@ export function buildRunManifest(opts: {
|
||||
}
|
||||
|
||||
const entries: ManifestEntry[] = [];
|
||||
const overlaySlice = opts.sliceCount;
|
||||
if (budgetMode) {
|
||||
const plan = packBySliceBudget(runnable.map(files => files[0]!), opts.sliceBudgetMs!, opts.jobs!,
|
||||
opts.durations ?? loadPaidTestDurations(rootDir, opts.tier), opts.timeoutMs);
|
||||
plan.slices.forEach((files, index) => files.forEach(file => entries.push({ file, slice: index + 1, status: 'planned',
|
||||
estimatedMs: plan.estimates[file]!,
|
||||
...(FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file)) ? { budget: resolvePaidShardBudget([file], opts.timeoutMs) } : {}) })));
|
||||
for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason });
|
||||
for (const e of excluded) entries.push({ file: e.file, slice: 0, status: 'excluded', reason: e.reason });
|
||||
entries.sort((a, b) => (a.file < b.file ? -1 : 1));
|
||||
return parseRunManifest(JSON.stringify({
|
||||
version: 1, tier: opts.tier, evalsAll: opts.evalsAll, sliceCount: plan.slices.length,
|
||||
selectionReason: cases.reason, profile, selection: cases.selection,
|
||||
...(cases.coverage ? { prCoverage: cases.coverage } : {}),
|
||||
plan: { sliceBudgetMs: opts.sliceBudgetMs!, jobs: opts.jobs!, estimatedSliceMs: plan.estimatedSliceMs, ciTimeoutMinutes: plan.ciTimeoutMinutes },
|
||||
entries,
|
||||
} satisfies PaidRunManifest));
|
||||
}
|
||||
const sliceCount = opts.sliceCount!;
|
||||
const overlaySlice = sliceCount;
|
||||
const reserveOverlaySlice = overlaySlice > 1 && runnable.some(files => files.some(isOverlayTestFile));
|
||||
const ordinarySlices = overlaySlice - Number(reserveOverlaySlice);
|
||||
// Spread registered long files by supervised load. Keep one ordinary-only
|
||||
// lane when possible, so every lane does not inherit a long-workflow tail.
|
||||
// The reserved overlay slice retains its ownership.
|
||||
const ordinary = runnable.filter(files => !files.some(isOverlayTestFile));
|
||||
const registered = ordinary.filter(files => FILE_RETRY_BUDGETS.some(budget => budget.file === files[0]));
|
||||
const registered = ordinary.filter(files => FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(files[0]!)));
|
||||
const allocations = new Map<string, number>();
|
||||
if (registered.length && ordinarySlices > 1) {
|
||||
const loads = Array<number>(ordinarySlices).fill(0);
|
||||
@@ -1220,7 +1467,7 @@ export function buildRunManifest(opts: {
|
||||
}
|
||||
const packed = packByRecordedDuration();
|
||||
function packByRecordedDuration(): Map<string, number> | null {
|
||||
const recorded = opts.durations ?? loadPaidTestDurations(rootDir);
|
||||
const recorded = opts.durations ?? loadPaidTestDurations(rootDir, opts.tier);
|
||||
if (ordinarySlices < 2 || ordinary.length === 0 || Object.keys(recorded).length === 0) return null;
|
||||
const bound = (files: string[], jobs: number) => paidShardWallUpperBoundMs([...files].sort(), jobs, opts.timeoutMs);
|
||||
const lanes = Array.from({ length: ordinarySlices }, (_, lane) =>
|
||||
@@ -1267,7 +1514,7 @@ export function buildRunManifest(opts: {
|
||||
runnable.forEach((files) => {
|
||||
const slice = files.some(isOverlayTestFile) ? overlaySlice : (packed ?? allocations).get(files[0])!;
|
||||
entries.push({ file: files[0], slice, status: 'planned',
|
||||
...(FILE_RETRY_BUDGETS.some(budget => budget.file === files[0])
|
||||
...(FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(files[0]!))
|
||||
? { budget: resolvePaidShardBudget(files, opts.timeoutMs) } : {}) });
|
||||
});
|
||||
for (const s of skipped) entries.push({ file: s.files[0], slice: 0, status: 'skipped-by-diff', reason: s.reason });
|
||||
@@ -1278,7 +1525,7 @@ export function buildRunManifest(opts: {
|
||||
version: 1,
|
||||
tier: opts.tier,
|
||||
evalsAll: opts.evalsAll,
|
||||
sliceCount: opts.sliceCount,
|
||||
sliceCount,
|
||||
selectionReason: cases.reason,
|
||||
profile,
|
||||
selection: cases.selection,
|
||||
@@ -1288,10 +1535,25 @@ export function buildRunManifest(opts: {
|
||||
return parseRunManifest(JSON.stringify(manifest));
|
||||
}
|
||||
|
||||
/**
|
||||
* Narrow a built manifest to a curated case subset (validation phases): the
|
||||
* selection binds every child, and a case shard outside it can execute
|
||||
* nothing, so it becomes skipped instead of an empty planned shard.
|
||||
*/
|
||||
export function restrictManifestSelection(manifest: PaidRunManifest, selection: PaidCaseSelection, reason: string): PaidRunManifest {
|
||||
const entries = manifest.entries.map(entry => {
|
||||
const caseId = shardCaseId(entry.file);
|
||||
if (entry.status !== 'planned' || caseId === null || selection.e2e === null || selection.e2e.includes(caseId)) return entry;
|
||||
const { estimatedMs: _estimate, budget: _budget, ...rest } = entry;
|
||||
return { ...rest, slice: 0, status: 'skipped-by-diff' as const, reason };
|
||||
});
|
||||
return parseRunManifest(JSON.stringify({ ...manifest, selection, entries }));
|
||||
}
|
||||
|
||||
export function parseRunManifest(raw: string): PaidRunManifest {
|
||||
const parsed = JSON.parse(raw) as PaidRunManifest;
|
||||
if (parsed.version !== 1) throw new Error(`unsupported manifest version: ${(parsed as { version?: unknown }).version}`);
|
||||
if (parsed.tier !== 'gate' && parsed.tier !== 'periodic') throw new Error(`manifest tier invalid: ${parsed.tier}`);
|
||||
if (!PAID_TIERS.includes(parsed.tier)) throw new Error(`manifest tier invalid: ${parsed.tier}`);
|
||||
if (parsed.profile !== undefined && parsed.profile !== 'pr' && parsed.profile !== 'full') throw new Error('manifest profile invalid');
|
||||
if (parsed.selection !== undefined) {
|
||||
for (const [key, inventory] of [['e2e', E2E_TOUCHFILES], ['judges', LLM_JUDGE_TOUCHFILES]] as const) {
|
||||
@@ -1329,19 +1591,50 @@ export function parseRunManifest(raw: string): PaidRunManifest {
|
||||
if (entry.status === 'planned' && (entry.slice < 1 || entry.slice > parsed.sliceCount)) {
|
||||
throw new Error(`planned entry ${entry.file} has out-of-range slice ${entry.slice}`);
|
||||
}
|
||||
if (entry.estimatedMs !== undefined && (entry.status !== 'planned' || !Number.isSafeInteger(entry.estimatedMs) || entry.estimatedMs < 0)) {
|
||||
throw new Error(`manifest entry ${entry.file} has an invalid estimate`);
|
||||
}
|
||||
if (entry.status === 'planned' && parsed.prCoverage?.mode === 'pr' && !prProfileFileSelected(entry.file, parsed.selection!)) {
|
||||
throw new Error(`manifest file is outside its PR case selection: ${entry.file}`);
|
||||
}
|
||||
}
|
||||
for (const entry of parsed.entries) {
|
||||
const caseId = shardCaseId(entry.file);
|
||||
if (caseId === null ? entry.status === 'planned' && CASE_SHARDED_FILES.includes(shardFile(entry.file))
|
||||
: !CASE_SHARDED_FILES.includes(shardFile(entry.file)) || !(caseId in E2E_TOUCHFILES)) {
|
||||
throw new Error(`Case-sharded files plan one registered case per shard: ${entry.file}`);
|
||||
}
|
||||
if (caseId !== null && entry.status === 'planned' && parsed.selection?.e2e && !parsed.selection.e2e.includes(caseId)) {
|
||||
throw new Error(`Planned case shard is outside the manifest selection: ${entry.file}`);
|
||||
}
|
||||
}
|
||||
if (parsed.prCoverage?.mode === 'pr') {
|
||||
const required = Object.entries(PR_PROFILE_FILES).filter(([, ids]) => ids.some(id => parsed.selection!.e2e!.includes(id))).map(([file]) => file);
|
||||
if (parsed.selection!.judges!.length) required.push('test/skill-llm-eval.test.ts');
|
||||
for (const file of required) {
|
||||
if (parsed.entries.filter(entry => entry.file === file && entry.status === 'planned').length !== 1) {
|
||||
throw new Error(`PR selected cases require exactly one planned owning file: ${file}`);
|
||||
const planned = parsed.entries.filter(entry => entry.status === 'planned').map(entry => normalizeRelativePath(entry.file));
|
||||
const required: string[][] = Object.entries(PR_PROFILE_FILES).flatMap(([file, ids]) => {
|
||||
const selected = ids.filter(id => parsed.selection!.e2e!.includes(id));
|
||||
if (!selected.length) return [];
|
||||
return [CASE_SHARDED_FILES.includes(file) ? selected.map(id => `${file}#${id}`) : [file]];
|
||||
});
|
||||
if (parsed.selection!.judges!.length) required.push(['test/skill-llm-eval.test.ts']);
|
||||
for (const owners of required) {
|
||||
if (owners.some(owner => planned.filter(key => key === owner).length !== 1)
|
||||
|| planned.filter(key => shardFile(key) === shardFile(owners[0]!)).length !== owners.length) {
|
||||
throw new Error(`PR selected cases require exactly one planned owning file: ${owners.join(', ')}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (parsed.plan !== undefined) {
|
||||
const plan = parsed.plan;
|
||||
const count = (value: unknown) => Number.isSafeInteger(value) && Number(value) > 0;
|
||||
if (!plan || typeof plan !== 'object' || !count(plan.sliceBudgetMs) || !count(plan.jobs) || !count(plan.ciTimeoutMinutes)
|
||||
|| !Array.isArray(plan.estimatedSliceMs) || plan.estimatedSliceMs.length !== parsed.sliceCount
|
||||
|| !plan.estimatedSliceMs.every(ms => Number.isSafeInteger(ms) && ms >= 0)
|
||||
|| parsed.entries.some(entry => entry.status === 'planned' && entry.estimatedMs === undefined)) {
|
||||
throw new Error('manifest slice plan malformed');
|
||||
}
|
||||
}
|
||||
const keys = parsed.entries.map(entry => normalizeRelativePath(entry.file));
|
||||
if (new Set(keys).size !== keys.length) throw new Error('Duplicate manifest entry');
|
||||
const overlaySlice = parsed.sliceCount;
|
||||
const plannedOverlays = parsed.entries.filter(entry => entry.status === 'planned' && isOverlayTestFile(entry.file));
|
||||
if (plannedOverlays.some(entry => entry.slice !== overlaySlice)) {
|
||||
@@ -1352,8 +1645,8 @@ export function parseRunManifest(raw: string): PaidRunManifest {
|
||||
throw new Error('The final ordinary manifest slice is reserved for overlay files');
|
||||
}
|
||||
for (const budget of FILE_RETRY_BUDGETS) {
|
||||
const entries = parsed.entries.filter(entry => normalizeRelativePath(entry.file) === budget.file);
|
||||
if (entries.length > 1) throw new Error(`Duplicate registered manifest entry: ${budget.file}`);
|
||||
const entries = parsed.entries.filter(entry => shardFile(entry.file) === budget.file);
|
||||
if (entries.length > 1 && entries.some(entry => shardCaseId(entry.file) === null)) throw new Error(`Duplicate registered manifest entry: ${budget.file}`);
|
||||
for (const entry of entries.filter(entry => entry.status === 'planned')) {
|
||||
if (!entry.budget) throw new Error(`Registered manifest needs an explicit budget record: ${budget.file}`);
|
||||
const expected = resolvePaidShardBudget([entry.file], entry.budget.source === 'explicit' ? entry.budget.timeoutMs : undefined);
|
||||
@@ -1371,7 +1664,7 @@ export interface SliceResult {
|
||||
sliceIndex: number;
|
||||
sliceCount: number;
|
||||
timeoutOverrideMs?: number;
|
||||
outcomes: Array<Pick<ShardOutcome, 'files' | 'status' | 'exitCode' | 'elapsedMs' | 'executedTests' | 'skippedTests' | 'budget'>>;
|
||||
outcomes: Array<Pick<ShardOutcome, 'files' | 'status' | 'exitCode' | 'elapsedMs' | 'executedTests' | 'skippedTests' | 'budget' | 'reused'>>;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1405,7 +1698,7 @@ export function verifySliceResults(
|
||||
const reported = new Map<string, { slice: number; status: ShardStatus }>();
|
||||
for (const result of results) {
|
||||
for (const outcome of result.outcomes) {
|
||||
if (outcome.files.some(file => FILE_RETRY_BUDGETS.some(budget => budget.file === normalizeRelativePath(file))) && outcome.files.length !== 1) {
|
||||
if (outcome.files.some(file => FILE_RETRY_BUDGETS.some(budget => budget.file === shardFile(file)) || shardCaseId(file) !== null) && outcome.files.length !== 1) {
|
||||
problems.push('Registered result must report its own shard');
|
||||
}
|
||||
const file = normalizeRelativePath(outcome.files[0] ?? '');
|
||||
@@ -1417,9 +1710,21 @@ export function verifySliceResults(
|
||||
problems.push(`PR profile expected ${expected} executed cases in ${file}, received ${executed}`);
|
||||
}
|
||||
}
|
||||
if (shardCaseId(file) !== null && outcome.status === 'passed'
|
||||
&& (outcome.executedTests === null || outcome.skippedTests === null || outcome.executedTests - outcome.skippedTests !== 1)) {
|
||||
problems.push(`Case shard must execute exactly its one case: ${file}`);
|
||||
}
|
||||
if (outcome.reused !== undefined) {
|
||||
const r = outcome.reused;
|
||||
if (manifest.prCoverage?.mode !== 'pr') problems.push(`${file}: only the fast PR profile may reuse results; this lane executes fresh`);
|
||||
if (outcome.status !== 'passed' || outcome.exitCode !== 0 || !/^[a-f0-9]{64}$/.test(r?.inputKey ?? '')
|
||||
|| !/^[\w./-]{1,160}$/.test(r?.runId ?? '') || !/^[a-f0-9]{40}$/.test(r?.revision ?? '') || !Number.isSafeInteger(r?.completedAt) || r.completedAt <= 0) {
|
||||
problems.push(`${file}: malformed reused result`);
|
||||
}
|
||||
}
|
||||
if (reported.has(file)) problems.push(`${file} reported by two slices`);
|
||||
reported.set(file, { slice: result.sliceIndex, status: outcome.status });
|
||||
const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === file);
|
||||
const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === shardFile(file));
|
||||
const finding = STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === file);
|
||||
if (finding) {
|
||||
// Full-census runs must account for every registered case. A manifest
|
||||
@@ -1453,6 +1758,22 @@ export function verifySliceResults(
|
||||
return { ok: problems.length === 0, problems };
|
||||
}
|
||||
|
||||
/** Budget-mode plan lines: every slice with its estimate, files and retries. */
|
||||
export function formatSlicePlan(manifest: PaidRunManifest): string[] {
|
||||
const plan = manifest.plan;
|
||||
if (!plan) return [];
|
||||
const minutes = (ms: number) => (ms / 60_000).toFixed(1);
|
||||
const lines = [`[test:paid] slice plan: ${manifest.sliceCount} slice(s) x ${plan.jobs} worker(s), budget ${minutes(plan.sliceBudgetMs)}m per slice, `
|
||||
+ `longest estimate ${minutes(Math.max(0, ...plan.estimatedSliceMs))}m, CI job timeout ${plan.ciTimeoutMinutes}m`];
|
||||
for (let slice = 1; slice <= manifest.sliceCount; slice++) {
|
||||
const mine = sliceExecutionOrder(manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice));
|
||||
const over = plan.estimatedSliceMs[slice - 1]! > plan.sliceBudgetMs ? ' [over budget: longer than one runner allows]' : '';
|
||||
lines.push(` slice ${slice}: ~${minutes(plan.estimatedSliceMs[slice - 1]!)}m${over}`);
|
||||
for (const entry of mine) lines.push(` ${entry.file} ~${minutes(entry.estimatedMs ?? 0)}m retries=${retriesForFiles([entry.file])}`);
|
||||
}
|
||||
return lines;
|
||||
}
|
||||
|
||||
export function formatProfileCoverage(manifest: PaidRunManifest): string[] {
|
||||
const coverage = manifest.prCoverage;
|
||||
return [
|
||||
@@ -1497,6 +1818,9 @@ type CliOptions = {
|
||||
emitPlanPath: string | null;
|
||||
/** Slice count for --emit-plan. */
|
||||
slices: number;
|
||||
/** Budget mode for --emit-plan / --list: per-executor estimated wall. */
|
||||
sliceBudgetMs: number | null;
|
||||
jobsExplicit: boolean;
|
||||
/** Executor mode: consume this manifest... */
|
||||
planPath: string | null;
|
||||
/** ...running only this 1-based slice. */
|
||||
@@ -1519,10 +1843,10 @@ function validatedTier(value: string | undefined, source: string): PaidTier {
|
||||
// otherwise cast through unchecked, match nothing in the runtime E2E_TIERS
|
||||
// filter, self-skip every test, and exit 0 with all shards 'passed' — the
|
||||
// exact 0%-execution-looks-like-a-pass class this runner exists to kill.
|
||||
if (value !== 'gate' && value !== 'periodic') {
|
||||
throw new Error(`${source} must be gate or periodic. Received: ${value}`);
|
||||
if (!PAID_TIERS.includes(value as PaidTier)) {
|
||||
throw new Error(`${source} must be gate, periodic or marathon. Received: ${value}`);
|
||||
}
|
||||
return value;
|
||||
return value as PaidTier;
|
||||
}
|
||||
|
||||
function validatedProfile(value: string | undefined, source: string): PaidProfile {
|
||||
@@ -1553,6 +1877,8 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process
|
||||
maxFilesPerShard: DEFAULT_MAX_FILES_PER_SHARD,
|
||||
emitPlanPath: null,
|
||||
slices: 1,
|
||||
sliceBudgetMs: null,
|
||||
jobsExplicit: !!env.EVALS_JOBS,
|
||||
planPath: null,
|
||||
sliceIndex: null,
|
||||
reportDir: null,
|
||||
@@ -1563,9 +1889,7 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process
|
||||
const arg = argv[index];
|
||||
if (arg === '--list') { options.listOnly = true; continue; }
|
||||
if (arg === '--tier') {
|
||||
const value = argv[index += 1];
|
||||
if (value !== 'gate' && value !== 'periodic') throw new Error(`--tier must be gate or periodic. Received: ${value}`);
|
||||
options.tier = value;
|
||||
options.tier = validatedTier(argv[index += 1] ?? '-', '--tier');
|
||||
continue;
|
||||
}
|
||||
if (arg === '--profile') {
|
||||
@@ -1574,7 +1898,8 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process
|
||||
options.profile = validatedProfile(value, '--profile'); options.profileExplicit = true; continue;
|
||||
}
|
||||
if (arg === '--timeout') { options.timeoutMs = parsePositiveInt(argv[index += 1], '--timeout') * 1000; options.timeoutExplicit = true; continue; }
|
||||
if (arg === '--jobs') { options.jobs = parsePositiveInt(argv[index += 1], '--jobs'); continue; }
|
||||
if (arg === '--jobs') { options.jobs = parsePositiveInt(argv[index += 1], '--jobs'); options.jobsExplicit = true; continue; }
|
||||
if (arg === '--slice-budget') { options.sliceBudgetMs = parsePositiveInt(argv[index += 1], '--slice-budget') * 1000; continue; }
|
||||
if (arg === '--files-per-shard') { options.maxFilesPerShard = parsePositiveInt(argv[index += 1], '--files-per-shard'); continue; }
|
||||
if (arg === '--emit-plan') {
|
||||
const value = argv[index += 1];
|
||||
@@ -1598,6 +1923,8 @@ export function parseCliOptions(argv: string[], env: NodeJS.ProcessEnv = process
|
||||
throw new Error(`Unknown argument: ${arg}`);
|
||||
}
|
||||
if (options.writeDurations && !options.reportDir) throw new Error('--write-durations requires --report');
|
||||
if (options.sliceBudgetMs !== null && argv.includes('--slices')) throw new Error('Plan with exactly one of --slices or --slice-budget');
|
||||
if (options.sliceBudgetMs !== null && !options.jobsExplicit) throw new Error('--slice-budget needs explicit --jobs (or EVALS_JOBS): the plan packs and supervises for that worker count');
|
||||
if (options.skipJudges && (!options.emitPlanPath || options.tier !== 'gate')) throw new Error('--skip-judges applies only to an emitted gate census plan');
|
||||
if (options.profile === 'pr' && options.tier !== 'gate') throw new Error('PR profile requires gate tier');
|
||||
if (options.profile === 'pr' && options.maxFilesPerShard !== 1) throw new Error('PR profile requires one file per shard to preserve case accounting');
|
||||
@@ -1613,7 +1940,7 @@ async function main(): Promise<number> {
|
||||
const manifest = buildRunManifest({
|
||||
tier: options.tier,
|
||||
profile: options.profile,
|
||||
sliceCount: options.slices,
|
||||
...(options.sliceBudgetMs !== null ? { sliceBudgetMs: options.sliceBudgetMs, jobs: options.jobs } : { sliceCount: options.slices }),
|
||||
timeoutMs: options.timeoutExplicit ? options.timeoutMs : undefined,
|
||||
evalsAll: process.env.EVALS_ALL === '1',
|
||||
skipJudges: options.skipJudges,
|
||||
@@ -1628,6 +1955,7 @@ async function main(): Promise<number> {
|
||||
+ `${planned} planned across ${manifest.sliceCount} slice(s), ${skipped} skipped by diff, `
|
||||
+ `${excludedCount} excluded (${manifest.selectionReason})`,
|
||||
);
|
||||
for (const line of formatSlicePlan(manifest)) console.log(line);
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -1647,16 +1975,14 @@ async function main(): Promise<number> {
|
||||
for (const line of formatProfileCoverage(manifest)) console.log(line);
|
||||
for (const result of results.sort((a, b) => a.sliceIndex - b.sliceIndex)) {
|
||||
for (const outcome of result.outcomes) {
|
||||
console.log(` slice ${result.sliceIndex} ${outcome.status.padEnd(15)} ${String(Math.round(outcome.elapsedMs / 1000)).padStart(5)}s ${outcome.files.join(' ')}`);
|
||||
const shown = outcome.reused ? `reused (run ${outcome.reused.runId})` : outcome.status;
|
||||
console.log(` slice ${result.sliceIndex} ${shown.padEnd(15)} ${String(Math.round(outcome.elapsedMs / 1000)).padStart(5)}s ${outcome.files.join(' ')}`);
|
||||
}
|
||||
}
|
||||
if (options.writeDurations) {
|
||||
const durations = mergePaidTestDurations(loadPaidTestDurations(), results);
|
||||
const target = path.join(ROOT, PAID_TEST_DURATIONS_FILE);
|
||||
const temporary = `${target}.tmp-${process.pid}`;
|
||||
fs.writeFileSync(temporary, `${JSON.stringify({ version: 1, recordedAt: new Date().toISOString(), durations }, null, 2)}\n`);
|
||||
fs.renameSync(temporary, target);
|
||||
console.log(`[test:paid] wrote ${Object.keys(durations).length} durations to ${PAID_TEST_DURATIONS_FILE}`);
|
||||
const durations = mergePaidTestDurations(loadPaidTestDurations(ROOT, manifest.tier), results);
|
||||
writePaidTestDurations(manifest.tier, durations);
|
||||
console.log(`[test:paid] wrote ${Object.keys(durations).length} ${manifest.tier} durations to ${PAID_TEST_DURATIONS_FILE}`);
|
||||
}
|
||||
// Historical flaky_retries includes every case with multiple attempts,
|
||||
// whether its final result passed or failed. Report attempts separately
|
||||
@@ -1767,7 +2093,10 @@ async function main(): Promise<number> {
|
||||
if (options.sliceIndex > manifest.sliceCount) {
|
||||
throw new Error(`--slice ${options.sliceIndex} exceeds manifest sliceCount ${manifest.sliceCount}`);
|
||||
}
|
||||
const mine = manifest.entries.filter((e) => e.status === 'planned' && e.slice === options.sliceIndex);
|
||||
if (manifest.plan && options.jobs !== manifest.plan.jobs) {
|
||||
throw new Error(`manifest was packed for ${manifest.plan.jobs} worker(s) per slice; EVALS_JOBS=${options.jobs} would break its supervision bound`);
|
||||
}
|
||||
const mine = sliceExecutionOrder(manifest.entries.filter((e) => e.status === 'planned' && e.slice === options.sliceIndex));
|
||||
const shards = mine.map((e) => [e.file]);
|
||||
for (const files of shards) resolvePaidShardTimeoutMs(files, timeoutOverride);
|
||||
console.log(`[test:paid] slice ${options.sliceIndex}/${manifest.sliceCount}: ${shards.length} shard(s), tier=${manifest.tier}, evalsAll=${manifest.evalsAll}`);
|
||||
@@ -1775,7 +2104,7 @@ async function main(): Promise<number> {
|
||||
if (options.listOnly) {
|
||||
for (const [index, files] of shards.entries()) {
|
||||
const budget = resolvePaidShardBudget(files, timeoutOverride);
|
||||
console.log(` shard ${index + 1}/${shards.length}: ${files.join(' ')} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'}`);
|
||||
console.log(` shard ${index + 1}/${shards.length}: ${files.join(' ')} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'} retries=${retriesForFiles(files)}`);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -1794,6 +2123,18 @@ async function main(): Promise<number> {
|
||||
...(manifest.prCoverage?.mode === 'pr' ? {
|
||||
expectedCases: Object.fromEntries(mine.map(entry => [entry.file, expectedPrCaseCount(entry.file, manifest.selection!)])),
|
||||
casePatterns: Object.fromEntries(mine.map(entry => [entry.file, prProfileTestNamePattern(entry.file, manifest.selection!)])),
|
||||
expectedCaseIds: Object.fromEntries(mine.map(entry => [entry.file, prProfileShardIds(entry.file, manifest.selection!)])),
|
||||
reuseFor: e2eReuseLaneProblem(process.env, manifest.prCoverage.mode) !== null ? undefined : (files, env, budget) => {
|
||||
const key = files[0]!;
|
||||
const file = shardFile(key);
|
||||
if (files.length !== 1 || !/^test\/skill-e2e-/.test(file)) return null;
|
||||
const { registered, known } = fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8'));
|
||||
return prepareE2EShardReuse({ root: ROOT, key, file, caseIds: prProfileShardIds(key, manifest.selection!),
|
||||
registeredIds: registered, registrationKnown: known,
|
||||
casePattern: prProfileTestNamePattern(key, manifest.selection!), expectedCases: expectedPrCaseCount(key, manifest.selection!),
|
||||
retries: retriesForFiles(files), timeoutMs: budget.timeoutMs, withinShardConcurrency: options.withinShardConcurrency,
|
||||
tier: manifest.tier, profile, env });
|
||||
},
|
||||
} : {}),
|
||||
env: {
|
||||
...process.env,
|
||||
@@ -1821,8 +2162,8 @@ async function main(): Promise<number> {
|
||||
sliceIndex: options.sliceIndex,
|
||||
sliceCount: manifest.sliceCount,
|
||||
...(options.timeoutExplicit ? { timeoutOverrideMs: options.timeoutMs } : {}),
|
||||
outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget }) =>
|
||||
({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}) })),
|
||||
outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused }) =>
|
||||
({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}), ...(reused ? { reused } : {}) })),
|
||||
};
|
||||
fs.mkdirSync(evalDirBase, { recursive: true });
|
||||
const sliceResultPath = path.join(evalDirBase, `slice-${options.sliceIndex}.json`);
|
||||
@@ -1832,8 +2173,16 @@ async function main(): Promise<number> {
|
||||
return summaryExitCode(summary);
|
||||
}
|
||||
|
||||
if (options.listOnly && options.sliceBudgetMs !== null) {
|
||||
const manifest = buildRunManifest({ tier: options.tier, profile: options.profile, sliceBudgetMs: options.sliceBudgetMs,
|
||||
jobs: options.jobs, evalsAll: process.env.EVALS_ALL === '1', timeoutMs: timeoutOverride });
|
||||
console.log(`[test:paid] slice plan preview: tier=${manifest.tier} profile=${manifest.profile ?? 'full'} (${manifest.selectionReason})`);
|
||||
for (const line of formatSlicePlan(manifest)) console.log(line);
|
||||
return 0;
|
||||
}
|
||||
|
||||
const { selected, excluded } = selectPaidTestFiles(discovered, options.tier);
|
||||
const shards = planPaidShards(selected, { maxFilesPerShard: options.maxFilesPerShard });
|
||||
const shards = planPaidShards(expandCaseShards(selected, options.tier), { maxFilesPerShard: options.maxFilesPerShard });
|
||||
|
||||
// Parent-side diff selection (D9): skip whole shards whose mapped tests are
|
||||
// all unselected. Fail-open everywhere — the child's self-skip stays
|
||||
@@ -1862,7 +2211,7 @@ async function main(): Promise<number> {
|
||||
const key = shards[index].join(' ');
|
||||
const note = skipReasons.has(key) ? ` [would skip: ${skipReasons.get(key)}]` : '';
|
||||
const budget = resolvePaidShardBudget(shards[index], options.timeoutExplicit ? options.timeoutMs : undefined);
|
||||
console.log(` shard ${index + 1}/${shards.length}: ${key} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'}${note}`);
|
||||
console.log(` shard ${index + 1}/${shards.length}: ${key} wall=${budget.timeoutMs}ms source=${budget.source} policy=${budget.policyId ?? 'none'} retries=${retriesForFiles(shards[index])}${note}`);
|
||||
}
|
||||
if (excluded.length > 0) {
|
||||
console.log(`\nExcluded (${excluded.length}):`);
|
||||
|
||||
@@ -5,7 +5,7 @@ import * as path from 'node:path';
|
||||
import { runPaidShard, shardSlug } from '../scripts/test-paid-shards';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const workflows = ['evals.yml', 'evals-periodic.yml'].map(name => ({
|
||||
const workflows = ['evals.yml', 'evals-periodic.yml', 'evals-marathon.yml'].map(name => ({
|
||||
name,
|
||||
value: Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows', name), 'utf8')) as any,
|
||||
}));
|
||||
@@ -23,7 +23,7 @@ function render(template: string, fields: Record<string, string>): string {
|
||||
|
||||
test('every direct CI paid executor binds a safe unique run/attempt/job/slice identity', () => {
|
||||
expect(executors.map(({ name, jobName }) => `${name}:${jobName}`)).toEqual([
|
||||
'evals.yml:eval-slices', 'evals-periodic.yml:eval-slices', 'evals-periodic.yml:gate-census',
|
||||
'evals.yml:eval-slices', 'evals-periodic.yml:eval-slices', 'evals-periodic.yml:gate-census', 'evals-marathon.yml:eval-slices',
|
||||
]);
|
||||
const ids = new Set<string>();
|
||||
for (const [workflowIndex, { job, step }] of executors.entries()) {
|
||||
@@ -32,9 +32,11 @@ test('every direct CI paid executor binds a safe unique run/attempt/job/slice id
|
||||
expect(env.EVALS_RUN_ID).toBeString();
|
||||
for (const run of ['36302678692', '36302678693']) {
|
||||
for (const attempt of ['1', '2']) {
|
||||
for (const slice of job.strategy.matrix.slice) {
|
||||
// The planner sizes the matrix; cover more slices than any live plan.
|
||||
expect(job.strategy.matrix.slice).toMatch(/^\$\{\{ fromJSON\(needs\.plan-slices\.outputs\.(?:[a-z]+_)?slices\) \}\}$/);
|
||||
for (let slice = 1; slice <= 64; slice++) {
|
||||
const id = render(env.EVALS_RUN_ID, {
|
||||
'github.run_id': `${run}${workflowIndex === 0 ? '0' : '1'}`,
|
||||
'github.run_id': `${run}${workflowIndex}`,
|
||||
'github.run_attempt': attempt, 'matrix.slice': String(slice),
|
||||
});
|
||||
expect(id).toMatch(/^[A-Za-z0-9_-]+$/);
|
||||
|
||||
@@ -30,8 +30,10 @@ test('the actual CI cookie repair planner executes only eight dependent cases wi
|
||||
expect(manifest.evalsAll).toBe(false);
|
||||
expect(manifest.selection).toEqual({ e2e: ['browse-basic', 'browse-snapshot', 'qa-quick', 'qa-only-no-fix', 'design-review-detector-shim-dom', 'diagram-triplet', 'canary-workflow', 'benchmark-workflow'], judges: [] });
|
||||
expect(manifest.entries.filter(entry => entry.status === 'planned').map(entry => entry.file).sort()).toEqual([
|
||||
'test/skill-e2e-bws.test.ts', 'test/skill-e2e-deploy.test.ts', 'test/skill-e2e-design.test.ts', 'test/skill-e2e-diagram.test.ts', 'test/skill-e2e-qa-workflow.test.ts',
|
||||
'test/skill-e2e-bws.test.ts', 'test/skill-e2e-deploy.test.ts', 'test/skill-e2e-design.test.ts#design-review-detector-shim-dom', 'test/skill-e2e-diagram.test.ts', 'test/skill-e2e-qa-workflow.test.ts',
|
||||
]);
|
||||
// The case-sharded design file runs only its one selected cookie case.
|
||||
expect(manifest.entries.filter(entry => entry.file.startsWith('test/skill-e2e-design.test.ts#') && entry.status === 'skipped-by-diff').length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
test('the existing quality and behavior phases retain their complete separate shard census', () => {
|
||||
@@ -42,7 +44,10 @@ test('the existing quality and behavior phases retain their complete separate sh
|
||||
expect(quality.evalsAll).toBe(true);
|
||||
expect(behavior.evalsAll).toBe(true);
|
||||
expect(qualityFiles).toHaveLength(1);
|
||||
expect(behaviorFiles).toHaveLength(45);
|
||||
// 44 files (first-task-scaffold registers no gate case, so the gate lane
|
||||
// skips it); the five case-sharded files contribute one shard per gate case.
|
||||
expect(new Set(behaviorFiles.map(file => file.split('#')[0])).size).toBe(44);
|
||||
expect(behaviorFiles).toHaveLength(62);
|
||||
expect(behaviorFiles).toEqual(expect.arrayContaining([
|
||||
'test/skill-e2e-qa-callers.test.ts',
|
||||
'test/skill-e2e-qa-functional-fix.test.ts',
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as os from 'node:os';
|
||||
import * as path from 'node:path';
|
||||
import {
|
||||
e2eReuseEnvironment, e2eReuseLaneProblem, e2eShardIdentity, e2eShardInputFiles, prepareE2EShardReuse,
|
||||
type E2EShardReuseRequest,
|
||||
} from '../scripts/e2e-shard-reuse';
|
||||
import { buildRunManifest, fileCaseRegistration, runPaidShard, verifySliceResults, type SliceResult } from '../scripts/test-paid-shards';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const FILE = 'test/skill-e2e-deploy.test.ts';
|
||||
const scratch = fs.mkdtempSync(path.join(os.tmpdir(), 'e2e-reuse-'));
|
||||
const bin = path.join(scratch, 'bin');
|
||||
fs.mkdirSync(bin);
|
||||
fs.writeFileSync(path.join(bin, 'claude'), '#!/bin/sh\necho "9.9.9 (Claude Code)"\n', { mode: 0o755 });
|
||||
|
||||
const laneEnv = (over: NodeJS.ProcessEnv = {}): NodeJS.ProcessEnv => ({
|
||||
PATH: `${bin}${path.delimiter}${process.env.PATH}`, HOME: scratch,
|
||||
EVALS_TIER: 'gate', EVALS_PROFILE: 'pr', EVALS: '1',
|
||||
EVALS_CACHE_DIR: path.join(scratch, 'cache'), EVALS_CACHE_REPOSITORY: 'garrytan/gstack', EVALS_CACHE_PR: '42',
|
||||
EVALS_CACHE_RUNTIME_ID: 'a'.repeat(64), GITHUB_RUN_ID: '1001', GITHUB_RUN_ATTEMPT: '1', ANTHROPIC_API_KEY: 'sk-fixture',
|
||||
GSTACK_CLAUDE_CLI_VERSION: '9.9.9 (Claude Code)',
|
||||
...over,
|
||||
});
|
||||
|
||||
function request(over: Partial<E2EShardReuseRequest> = {}): E2EShardReuseRequest {
|
||||
const { registered, known } = fileCaseRegistration(FILE, fs.readFileSync(path.join(ROOT, FILE), 'utf8'));
|
||||
return { root: ROOT, key: FILE, file: FILE, caseIds: ['setup-deploy-workflow'], registeredIds: registered, registrationKnown: known,
|
||||
casePattern: '(?:^|\\s)(?:setup-deploy-workflow)$', expectedCases: 1, retries: 0, timeoutMs: 1_800_000,
|
||||
withinShardConcurrency: 2, tier: 'gate', profile: 'pr', env: laneEnv(), ...over };
|
||||
}
|
||||
|
||||
describe('E2E shard reuse eligibility', () => {
|
||||
test('only the same-PR fast profile with an immutable runtime and default endpoint may reuse', () => {
|
||||
expect(e2eReuseLaneProblem(laneEnv(), 'pr')).toBeNull();
|
||||
for (const [env, mode, problem] of [
|
||||
[laneEnv(), 'full-fallback', 'Only the fast PR profile'],
|
||||
[laneEnv(), undefined, 'Only the fast PR profile'],
|
||||
[laneEnv({ EVALS_CACHE_PR: '' }), 'pr', 'same-PR cache scope'],
|
||||
[laneEnv({ EVALS_CACHE_RUNTIME_ID: 'latest' }), 'pr', 'immutable runtime'],
|
||||
[laneEnv({ EVALS_FRESH: '1' }), 'pr', 'Fresh validation'],
|
||||
[laneEnv({ EVALS_TIER: 'periodic' }), 'pr', 'Fresh validation'],
|
||||
[laneEnv({ EVALS_CACHE_PURPOSE: 'periodic' }), 'pr', 'execute fresh'],
|
||||
[laneEnv({ EVALS_CACHE_PURPOSE: 'marathon' }), 'pr', 'execute fresh'],
|
||||
[laneEnv({ EVALS_CACHE_PURPOSE: 'release' }), 'pr', 'execute fresh'],
|
||||
[laneEnv({ NODE_OPTIONS: '--require x' }), 'pr', 'Preload'],
|
||||
[laneEnv({ ANTHROPIC_BASE_URL: 'https://proxy.example' }), 'pr', 'Custom model endpoint'],
|
||||
] as const) expect(e2eReuseLaneProblem(env, mode)).toContain(problem);
|
||||
});
|
||||
|
||||
test('the identity binds the child environment except run-scoped transport, and never secret values', () => {
|
||||
const env = e2eReuseEnvironment(laneEnv({ EVALS_RUN_ID: 'run-1', GSTACK_EVAL_DIR: '/tmp/x', EVALS_SELECTION_JSON: '{}', EVALS_MODEL: 'm', UNRELATED: 'x' }));
|
||||
expect(env.ANTHROPIC_API_KEY).toBe('set');
|
||||
expect(env.EVALS_MODEL).toBe('m');
|
||||
for (const name of ['EVALS_RUN_ID', 'GSTACK_EVAL_DIR', 'EVALS_SELECTION_JSON', 'EVALS_CACHE_DIR', 'EVALS_CACHE_PR', 'UNRELATED', 'GITHUB_RUN_ID']) {
|
||||
expect(env[name], name).toBeUndefined();
|
||||
}
|
||||
expect(JSON.stringify(env)).not.toContain('sk-fixture');
|
||||
});
|
||||
|
||||
test('consumed files cover the test closure, every registered touchfile, the globals and the harness', () => {
|
||||
const files = e2eShardInputFiles(request());
|
||||
for (const file of [FILE, 'test/helpers/e2e-helpers.ts', 'scripts/test-paid-shards.ts', 'scripts/e2e-shard-reuse.ts',
|
||||
'bun.lock', '.github/workflows/evals.yml', '.github/actions/register-gstack-skills/action.yml', '.github/docker/Dockerfile.ci',
|
||||
'setup-deploy/SKILL.md.tmpl', 'test/helpers/touchfiles-data.ts']) expect(files, file).toContain(file);
|
||||
expect(files).not.toContain('package.json');
|
||||
expect(files.some(file => file.startsWith('node_modules/'))).toBe(true);
|
||||
expect(() => e2eShardInputFiles({ ...request(), registeredIds: ['no-such-case'] })).not.toThrow();
|
||||
});
|
||||
|
||||
test('unknown or unprovable inputs fail closed', () => {
|
||||
expect(e2eShardIdentity(request()).status).toBe('eligible');
|
||||
for (const [over, reason] of [
|
||||
[{ retries: 1 }, 'first attempt'],
|
||||
[{ registrationKnown: false }, 'statically complete'],
|
||||
[{ caseIds: [] }, 'exactly known'],
|
||||
[{ expectedCases: 2 }, 'exactly known'],
|
||||
[{ caseIds: ['not-registered'] }, 'exactly known'],
|
||||
[{ file: 'test/skill-llm-eval.test.ts', key: 'test/skill-llm-eval.test.ts' }, 'audited E2E file'],
|
||||
[{ env: laneEnv({ PATH: path.join(scratch, 'empty') }) }, 'Claude CLI version is unknown'],
|
||||
] as const) {
|
||||
const result = e2eShardIdentity(request(over as Partial<E2EShardReuseRequest>));
|
||||
expect(result.status, reason).toBe('ineligible');
|
||||
expect(result.status === 'ineligible' ? result.reason : '').toContain(reason);
|
||||
}
|
||||
});
|
||||
|
||||
test('any consumed parameter, pin or runtime change is a different identity', () => {
|
||||
const key = (over: Partial<E2EShardReuseRequest>) => {
|
||||
const result = e2eShardIdentity(request(over));
|
||||
if (result.status !== 'eligible') throw new Error(result.reason);
|
||||
return result.identity.key;
|
||||
};
|
||||
const base = key({});
|
||||
expect(key({})).toBe(base);
|
||||
expect(key({ env: laneEnv({ EVALS_RUN_ID: 'another-run', GITHUB_RUN_ID: '9' }) })).toBe(base);
|
||||
for (const over of [{ timeoutMs: 1_000 }, { withinShardConcurrency: 1 }, { casePattern: 'x' },
|
||||
{ env: laneEnv({ EVALS_MODEL: 'other' }) }, { env: laneEnv({ EVALS_CACHE_RUNTIME_ID: 'b'.repeat(64) }) },
|
||||
{ env: laneEnv({ EVALS_CACHE_PR: '43' }) }] as Array<Partial<E2EShardReuseRequest>>) expect(key(over)).not.toBe(base);
|
||||
});
|
||||
});
|
||||
|
||||
describe('E2E shard reuse through the runner', () => {
|
||||
test('a fresh first-attempt pass publishes; identical inputs then reuse without launching; changed inputs run', async () => {
|
||||
const env = laneEnv({ EVALS_CACHE_DIR: path.join(scratch, 'roundtrip') });
|
||||
const first = prepareE2EShardReuse(request({ env }))!;
|
||||
expect(first.lookup()).toBeNull();
|
||||
first.publish();
|
||||
const hit = prepareE2EShardReuse(request({ env }))!.lookup();
|
||||
expect(hit?.source.runId).toBe('1001/1');
|
||||
expect(prepareE2EShardReuse(request({ env: { ...env, EVALS_MODEL: 'changed' } }))!.lookup()).toBeNull();
|
||||
expect(prepareE2EShardReuse(request({ env: { ...env, EVALS_FRESH: '1' } }))).toBeNull();
|
||||
|
||||
const evalDir = path.join(scratch, 'evals');
|
||||
let launched = 0;
|
||||
const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, evalDirBase: evalDir, env, log: () => {},
|
||||
expectedCaseIds: { [FILE]: ['setup-deploy-workflow'] },
|
||||
reuseFor: (files, childEnv) => prepareE2EShardReuse(request({ env: { ...childEnv } })),
|
||||
commandFor: () => { launched++; return { command: process.execPath, args: ['-e', 'process.exit(1)'] }; } });
|
||||
expect(launched).toBe(0);
|
||||
expect(outcome).toMatchObject({ status: 'passed', exitCode: 0, executedTests: 1, skippedTests: 0, reused: { runId: '1001/1' } });
|
||||
const recorded = JSON.parse(fs.readFileSync(path.join(evalDir, 'shards', 'skill-e2e-deploy', 'e2e-reused-skill-e2e-deploy.json'), 'utf8'));
|
||||
expect(recorded.tests).toEqual([expect.objectContaining({ name: 'setup-deploy-workflow', passed: true, execution: 'reused' })]);
|
||||
});
|
||||
|
||||
test('a failed shard never publishes a receipt', async () => {
|
||||
let published = 0;
|
||||
const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, env: laneEnv(), log: () => {},
|
||||
reuseFor: () => ({ lookup: () => null, publish: () => { published++; } }),
|
||||
commandFor: () => ({ command: process.execPath, args: ['-e', 'process.exit(1)'] }) });
|
||||
expect(outcome.status).toBe('failed');
|
||||
expect(published).toBe(0);
|
||||
});
|
||||
|
||||
test('the report accepts reused results only in the fast PR profile', () => {
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
const planned = manifest.entries.filter(entry => entry.status === 'planned');
|
||||
const reused = { inputKey: 'c'.repeat(64), runId: '1001/1', revision: 'd'.repeat(40), completedAt: 1 };
|
||||
const results: SliceResult[] = [{ version: 1, tier: 'gate', sliceIndex: 1, sliceCount: 1, outcomes: planned.map(entry => ({
|
||||
files: [entry.file], status: 'passed' as const, exitCode: 0, elapsedMs: 0, executedTests: 1, skippedTests: 0,
|
||||
...(entry.budget ? { budget: entry.budget } : {}), ...(entry.file === FILE ? { reused } : {}) })) }];
|
||||
expect(verifySliceResults(manifest, results).problems).toContain(`${FILE}: only the fast PR profile may reuse results; this lane executes fresh`);
|
||||
});
|
||||
});
|
||||
@@ -184,8 +184,8 @@ describe('E2E tier alignment (touchfiles declaration vs test self-gate)', () =>
|
||||
// Both self-gate shapes count: the raw predicate and the consolidated
|
||||
// helper (test/helpers/e2e-gate.ts documents this file as a consumer
|
||||
// that must recognize describeE2ETier/e2eTierEnabled).
|
||||
const selfGated = /EVALS_TIER\s*===\s*['"](gate|periodic)['"]/.test(content)
|
||||
|| /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic)['"]/.test(content);
|
||||
const selfGated = /EVALS_TIER\s*===\s*['"](gate|periodic|marathon)['"]/.test(content)
|
||||
|| /\b(?:describeE2ETier|e2eTierEnabled)\(\s*['"](gate|periodic|marathon)['"]/.test(content);
|
||||
if (!usesNameSelection && selfGated) continue; // fail-open-safe standalone
|
||||
|
||||
invisible.push(
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { expect, test } from 'bun:test';
|
||||
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, OVERLAY_MAX_ACTIVE_SHARDS, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS } from '../scripts/test-paid-shards';
|
||||
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards';
|
||||
import { FINDING_RETRY_BUDGETS, ALL_TIERS, SHARD_RESERVE_MS } from './helpers/eval-budgets';
|
||||
import fs from 'node:fs';
|
||||
import os from 'node:os';
|
||||
@@ -8,7 +8,8 @@ import path from 'node:path';
|
||||
for (const budget of FINDING_RETRY_BUDGETS) {
|
||||
test(`${budget.file}: supervision preserves every existing attempt and retry`, () => {
|
||||
expect(budget.testMs).toBe(1_500_000);
|
||||
expect(budget.retries).toBe(1);
|
||||
// A 25-minute case is past RETRY_MAX_CASE_MS: a timed-out attempt is its verdict.
|
||||
expect(budget.retries).toBe(0);
|
||||
expect(retriesForFiles([budget.file])).toBe(budget.retries);
|
||||
expect(budget.shardReserveMs).toBe(SHARD_RESERVE_MS);
|
||||
expect(budget.shardMs).toBe(budget.cases * budget.testMs * (budget.retries + 1) + budget.shardReserveMs);
|
||||
@@ -24,9 +25,8 @@ for (const budget of FINDING_RETRY_BUDGETS) {
|
||||
expect([...source.matchAll(/timeoutMs:\s*1_500_000\b/g)]).toHaveLength(budget.cases);
|
||||
}
|
||||
expect([...source.matchAll(/1_500_000\s*\/\* physical ceiling:/g)]).toHaveLength(budget.cases);
|
||||
// Current periodic CI already supports this supervision wall.
|
||||
const workflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any;
|
||||
expect(budget.shardMs).toBeLessThan(workflow.jobs['eval-slices']['timeout-minutes'] * 60_000);
|
||||
// The planned periodic CI job cap supports this supervision wall.
|
||||
expect(budget.shardMs).toBeLessThan(livePlan().plan!.ciTimeoutMinutes * 60_000);
|
||||
});
|
||||
|
||||
test(`${budget.file}: own-shard allocation leaves ordinary and explicit limits intact`, () => {
|
||||
@@ -104,31 +104,36 @@ test('actual shard launcher honors the explicit saved planner limit without a pr
|
||||
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
||||
}, 10000);
|
||||
|
||||
const periodicWorkflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any;
|
||||
const periodicJob = periodicWorkflow.jobs['eval-slices'];
|
||||
const periodicPlanStep = periodicWorkflow.jobs['plan-slices'].steps.find((step: any) => step.run?.includes('--tier periodic --emit-plan'));
|
||||
const periodicSliceCount = Number(periodicPlanStep.run.match(/--slices\s+(\d+)/)?.[1]);
|
||||
const periodicRunStep = periodicJob.steps.find((step: any) => step.run?.includes('--plan /tmp/paid-plan/manifest.json'));
|
||||
const periodicWorkers = Number(periodicRunStep.env.EVALS_JOBS);
|
||||
const livePlan = (discovered?: string[]) => buildRunManifest({ tier: 'periodic', sliceCount: periodicSliceCount,
|
||||
evalsAll: true, env: { EVALS_ALL: '1' }, discovered });
|
||||
function periodicLane() {
|
||||
const workflow = Bun.YAML.parse(fs.readFileSync(path.join(import.meta.dir, '../.github/workflows/evals-periodic.yml'), 'utf8')) as any;
|
||||
const job = workflow.jobs['eval-slices'];
|
||||
const planStep = workflow.jobs['plan-slices'].steps.find((step: any) => step.run?.includes('--tier periodic --emit-plan'));
|
||||
const planned = parseCliOptions(planStep.run.slice(planStep.run.indexOf('scripts/test-paid-shards.ts') + 'scripts/test-paid-shards.ts'.length).trim().split(/\s+/), {});
|
||||
const runStep = job.steps.find((step: any) => step.run?.includes('--plan /tmp/paid-plan/manifest.json'));
|
||||
return { job, planStep, planned, workers: Number(runStep.env.EVALS_JOBS) };
|
||||
}
|
||||
function livePlan(discovered?: string[]) {
|
||||
const { planned } = periodicLane();
|
||||
return buildRunManifest({ tier: 'periodic', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs,
|
||||
evalsAll: true, env: { EVALS_ALL: '1' }, discovered });
|
||||
}
|
||||
|
||||
test('live periodic census fits the declared CI wall including setup', () => {
|
||||
const { job, planStep, planned, workers } = periodicLane();
|
||||
const m = livePlan();
|
||||
expect(periodicPlanStep.run).not.toContain('--autoplan-slice');
|
||||
expect(periodicJob.strategy.matrix.slice).toEqual(Array.from({ length: periodicSliceCount }, (_, index) => index + 1));
|
||||
expect(periodicWorkers).toBe(2);
|
||||
const walls = Array.from({ length: periodicSliceCount }, (_, index) => {
|
||||
const files = m.entries.filter(e => e.status === 'planned' && e.slice === index + 1).map(e => e.file);
|
||||
const workers = files.some(isOverlayTestFile) ? Math.min(periodicWorkers, OVERLAY_MAX_ACTIVE_SHARDS) : periodicWorkers;
|
||||
return paidShardWallUpperBoundMs(files, workers);
|
||||
});
|
||||
expect(Math.max(...walls)).toBe(14_680_000);
|
||||
expect(periodicJob['timeout-minutes']).toBe(360);
|
||||
expect(periodicJob.strategy['max-parallel']).toBe(8);
|
||||
expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(periodicJob['timeout-minutes'] * 60_000);
|
||||
expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(70);
|
||||
const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount);
|
||||
expect(planStep.run).not.toContain('--autoplan-slice');
|
||||
expect(job.strategy.matrix.slice).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}');
|
||||
expect(job['timeout-minutes']).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}');
|
||||
expect(workers).toBe(2);
|
||||
expect(planned.jobs).toBe(workers);
|
||||
const walls = Array.from({ length: m.sliceCount }, (_, index) => sliceSupervisedWallMs(sliceExecutionOrder(
|
||||
m.entries.filter(e => e.status === 'planned' && e.slice === index + 1)).map(e => e.file), workers));
|
||||
expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(m.plan!.ciTimeoutMinutes * 60_000);
|
||||
expect(m.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
|
||||
expect(m.sliceCount).toBeLessThanOrEqual(job.strategy['max-parallel']);
|
||||
const plannedFiles = new Set(m.entries.filter(e => e.status === 'planned').map(e => shardFile(e.file)));
|
||||
expect(plannedFiles).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected));
|
||||
const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === m.sliceCount);
|
||||
expect(overlays).toHaveLength(4);
|
||||
expect(overlays.every(e => isOverlayTestFile(e.file))).toBe(true);
|
||||
});
|
||||
@@ -139,8 +144,8 @@ test('registered allocation is deterministic and preserves every discovered file
|
||||
expect(files).toContain('test/skill-e2e-ship-skip.test.ts');
|
||||
const m = livePlan(files);
|
||||
expect(livePlan([...files].reverse())).toEqual(m);
|
||||
expect(m.entries.map(e => e.file).sort()).toEqual([...files].sort());
|
||||
expect(new Set(m.entries.map(e => e.file)).size).toBe(files.length);
|
||||
expect([...new Set(m.entries.map(e => shardFile(e.file)))].sort()).toEqual([...files].sort());
|
||||
expect(new Set(m.entries.map(e => e.file)).size).toBe(m.entries.length);
|
||||
});
|
||||
|
||||
test('ordinary-only manifests retain round-robin allocation', () => {
|
||||
@@ -171,17 +176,18 @@ test('single-slice manifest retains all registered files with one allocation', (
|
||||
|
||||
test('current detach supervision covers the live-census floor', () => {
|
||||
const floorFor = (tier: 'gate' | 'periodic') => {
|
||||
const files = selectPaidTestFiles(collectPaidTestFiles(), tier).selected;
|
||||
// Case-sharded files contribute one shard per case, exactly as the runner plans.
|
||||
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier);
|
||||
const excess = files.reduce((n, file) => n + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - DEFAULT_SHARD_TIMEOUT_MS), 0);
|
||||
return Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05);
|
||||
};
|
||||
const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8'));
|
||||
const periodicTimeout = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]);
|
||||
const gateTimeout = Number(pkg.scripts['eval:bg:gate'].match(/--timeout\s+(\d+)/)[1]);
|
||||
expect(floorFor('gate')).toBe(42_851);
|
||||
expect(floorFor('gate')).toBe(26_597);
|
||||
expect(gateTimeout).toBe(49_320);
|
||||
expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate'));
|
||||
expect(floorFor('periodic')).toBe(37_727);
|
||||
expect(floorFor('periodic')).toBe(30_797);
|
||||
});
|
||||
|
||||
for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => {
|
||||
|
||||
@@ -24,9 +24,10 @@ import {
|
||||
DEFAULT_JOBS,
|
||||
DEFAULT_SHARD_TIMEOUT_MS,
|
||||
resolvePaidShardBudget,
|
||||
expandCaseShards,
|
||||
type PaidTier,
|
||||
} from '../scripts/test-paid-shards';
|
||||
import { FINDING_RETRY_BUDGETS } from './helpers/eval-budgets';
|
||||
import { FILE_RETRY_BUDGETS } from './helpers/eval-budgets';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
// 5% margin over the theoretical bound: detach setup, lock wait, aggregation.
|
||||
@@ -57,7 +58,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
|
||||
['periodic', 'eval:bg:periodic'],
|
||||
] as Array<[PaidTier, string]>) {
|
||||
test(`${script} covers ordinary ${tier} waves plus registered excess x ${MARGIN}`, () => {
|
||||
const files = selectPaidTestFiles(collectPaidTestFiles(), tier).selected;
|
||||
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier);
|
||||
expect(files.length).toBeGreaterThan(0);
|
||||
const floor = Math.ceil(worstCaseSeconds(files) * MARGIN);
|
||||
const configured = detachTimeoutSeconds(script);
|
||||
@@ -78,7 +79,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
|
||||
const pkg = JSON.parse(fs.readFileSync(path.join(ROOT, 'package.json'), 'utf8'));
|
||||
const jobs = Number(pkg.scripts['test:pr'].match(/EVALS_JOBS=\$\{EVALS_JOBS:-(\d+)\}/)?.[1]);
|
||||
expect(jobs).toBe(2);
|
||||
const files = selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected;
|
||||
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected, 'gate');
|
||||
const floor = Math.ceil(worstCaseSeconds(files, jobs) * MARGIN);
|
||||
expect(detachTimeoutSeconds('eval:bg:pr')).toBeGreaterThanOrEqual(floor);
|
||||
});
|
||||
@@ -86,7 +87,7 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
|
||||
test('eval:bg:release covers both complete tiers and their existing margins', () => {
|
||||
const files = collectPaidTestFiles();
|
||||
const floor = (['gate', 'periodic'] as const).reduce((sum, tier) =>
|
||||
sum + Math.ceil(worstCaseSeconds(selectPaidTestFiles(files, tier).selected) * MARGIN), 0);
|
||||
sum + Math.ceil(worstCaseSeconds(expandCaseShards(selectPaidTestFiles(files, tier).selected, tier)) * MARGIN), 0);
|
||||
expect(detachTimeoutSeconds('eval:bg:release')).toBeGreaterThanOrEqual(floor);
|
||||
});
|
||||
});
|
||||
@@ -94,7 +95,9 @@ describe('eval:bg detach timeouts cover the sharded runner worst case', () => {
|
||||
// One long job and one ordinary job can run side by side; the long job still
|
||||
// needs its whole wall, regardless of the number of ordinary workers.
|
||||
test('a heterogeneous pair rejects the old uniform-wall floor', () => {
|
||||
const pair = [FINDING_RETRY_BUDGETS[0]!.file, 'test/skill-e2e-other.test.ts'];
|
||||
// The longest registered wall (single-attempt finding files now fit the ordinary wall).
|
||||
const longest = [...FILE_RETRY_BUDGETS].sort((a, b) => b.shardMs - a.shardMs)[0]!;
|
||||
const pair = [longest.file, 'test/skill-e2e-other.test.ts'];
|
||||
const actualLongest = Math.max(...pair.map(file => resolvePaidShardBudget([file]).timeoutMs)) / 1000;
|
||||
expect(worstCaseSeconds(pair, 2)).toBe(actualLongest);
|
||||
expect(worstCaseSeconds(pair, 2)).toBeGreaterThan(DEFAULT_SHARD_TIMEOUT_MS / 1000);
|
||||
|
||||
@@ -22,25 +22,53 @@
|
||||
import { describe, test, expect } from 'bun:test';
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import { buildRunManifest, parseCliOptions, isOverlayTestFile, OVERLAY_MAX_ACTIVE_SHARDS, paidShardWallUpperBoundMs } from '../scripts/test-paid-shards';
|
||||
import { buildRunManifest, parseCliOptions, sliceExecutionOrder, sliceSupervisedWallMs, CI_SETUP_ALLOWANCE_MINUTES } from '../scripts/test-paid-shards';
|
||||
|
||||
const ROOT = path.join(import.meta.dir, '..');
|
||||
const read = (rel: string) => fs.readFileSync(path.join(ROOT, rel), 'utf-8');
|
||||
|
||||
const evalsYml = read('.github/workflows/evals.yml');
|
||||
const periodicYml = read('.github/workflows/evals-periodic.yml');
|
||||
const marathonYml = read('.github/workflows/evals-marathon.yml');
|
||||
const registerAction = read('.github/actions/register-gstack-skills/action.yml');
|
||||
|
||||
/** Slice count the planner emits (`--slices N`) in a workflow source. */
|
||||
function plannedSlices(source: string): number[] {
|
||||
return [...source.matchAll(/--emit-plan\s+\S+\s+--slices\s+(\d+)/g)].map((m) => Number(m[1]));
|
||||
/** Every planner site: its manifest path and budget (`--slice-budget S --jobs J`). */
|
||||
function plannerSites(source: string): Array<{ manifest: string; budgetSeconds: number; jobs: number }> {
|
||||
return [...source.matchAll(/--emit-plan\s+(\S+)\s+--slice-budget\s+(\d+)\s+--jobs\s+(\d+)/g)]
|
||||
.map((m) => ({ manifest: m[1]!, budgetSeconds: Number(m[2]), jobs: Number(m[3]) }));
|
||||
}
|
||||
|
||||
/** The executor matrix's slice list (`slice: [1, 2, ...]`). */
|
||||
function matrixSlices(source: string): number[][] {
|
||||
return [...source.matchAll(/^\s+slice: \[([\d,\s]+)\]\s*$/gm)].map((m) =>
|
||||
m[1].split(',').map((n) => Number(n.trim())),
|
||||
);
|
||||
type Step = { id?: string; name?: string; run?: string; env?: Record<string, string>; with?: Record<string, string> };
|
||||
type Job = { needs?: string[]; env?: Record<string, string>; outputs?: Record<string, string>; 'timeout-minutes': string | number;
|
||||
strategy?: { 'max-parallel': number; matrix: { slice: string } }; steps: Step[] };
|
||||
|
||||
/**
|
||||
* An executor's matrix and timeout must come from the planner step that wrote
|
||||
* the manifest it downloads: `slices` from `[range(1; .sliceCount + 1)]` and
|
||||
* `timeout-minutes` from `.plan.ciTimeoutMinutes`, never hand-written numbers.
|
||||
*/
|
||||
function expectPlannedExecutor(source: string, executorName: string, prefix: string) {
|
||||
const workflow = Bun.YAML.parse(source) as { jobs: Record<string, Job> };
|
||||
const planner = workflow.jobs['plan-slices']!;
|
||||
const executor = workflow.jobs[executorName]!;
|
||||
expect(executor.needs).toContain('plan-slices');
|
||||
expect(executor.strategy!.matrix.slice).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}slices) }}`);
|
||||
expect(executor['timeout-minutes']).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}timeout_minutes) }}`);
|
||||
const [stepId] = /^\$\{\{ steps\.([\w-]+)\.outputs\.slices \}\}$/.exec(planner.outputs![`${prefix}slices`]!)!.slice(1);
|
||||
expect(planner.outputs![`${prefix}timeout_minutes`]).toBe(`\${{ steps.${stepId}.outputs.timeout_minutes }}`);
|
||||
const matrixStep = planner.steps.find(step => step.id === stepId)!;
|
||||
const manifest = /jq -c '\[range\(1; \.sliceCount \+ 1\)\]' (\S+)\)/.exec(matrixStep.run!)![1]!;
|
||||
expect(matrixStep.run).toContain(`jq -e '.plan.ciTimeoutMinutes' ${manifest})`);
|
||||
const emit = planner.steps.filter(step => step.run?.includes(`--emit-plan ${manifest} `));
|
||||
expect(emit).toHaveLength(1);
|
||||
const execute = executor.steps.filter(step => step.run?.includes('--plan '));
|
||||
expect(execute).toHaveLength(1);
|
||||
expect(execute[0]!.run).toContain(`--plan ${manifest} --slice \${{ matrix.slice }}`);
|
||||
expect(executor.steps.some(step => step.with?.path === manifest.replace(/\/manifest\.json$/, ''))).toBe(true);
|
||||
// The planner packs for exactly the executor's worker count.
|
||||
const site = plannerSites(emit[0]!.run!)[0]!;
|
||||
expect(execute[0]!.env?.EVALS_JOBS).toBe(String(site.jobs));
|
||||
return { site, emit: emit[0]!, execute: execute[0]!, executor, planner };
|
||||
}
|
||||
|
||||
describe('evals.yml sliced-lane wiring (post-matrix)', () => {
|
||||
@@ -67,13 +95,12 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => {
|
||||
expect(evalsYml).toMatch(/EVALS_TIER=gate bun --no-install run scripts\/test-paid-shards\.ts --tier gate --report /);
|
||||
});
|
||||
|
||||
test('executor matrix slice list matches the planner --slices count', () => {
|
||||
const planned = plannedSlices(evalsYml);
|
||||
const matrices = matrixSlices(evalsYml);
|
||||
expect(planned, 'expected exactly one --emit-plan site in evals.yml').toHaveLength(1);
|
||||
expect(matrices, 'expected exactly one slice matrix in evals.yml').toHaveLength(1);
|
||||
const n = planned[0];
|
||||
expect(matrices[0]).toEqual(Array.from({ length: n }, (_, i) => i + 1));
|
||||
test('executor matrix and timeout come from the one budget planner', () => {
|
||||
expect(plannerSites(evalsYml), 'expected exactly one --emit-plan site in evals.yml').toHaveLength(1);
|
||||
const { site } = expectPlannedExecutor(evalsYml, 'eval-slices', '');
|
||||
expect(site).toEqual({ manifest: '/tmp/paid-plan/manifest.json', budgetSeconds: 540, jobs: 2 });
|
||||
// The validation-phase planner writes the same manifest with the same budget.
|
||||
expect(evalsYml).toContain('sliceBudgetMs: 540000, jobs: 2');
|
||||
});
|
||||
|
||||
test('reconcile exit is captured via PIPESTATUS, never $? after a pipe', () => {
|
||||
@@ -81,7 +108,7 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => {
|
||||
// `$?` after `... | tee` is tee's exit — always 0. That made the
|
||||
// fail-closed reconcile gate silently fail-open (ship review army,
|
||||
// 2026-08-31). Both lanes must read PIPESTATUS[0].
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) {
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) {
|
||||
const reconcileBlocks = [...source.matchAll(/--report[^\n]*\| tee[^\n]*\n([\s\S]{0,400}?)GITHUB_OUTPUT/g)];
|
||||
expect(reconcileBlocks.length, `${name}: expected a tee'd reconcile step`).toBeGreaterThanOrEqual(1);
|
||||
for (const block of reconcileBlocks) {
|
||||
@@ -124,78 +151,89 @@ describe('evals.yml sliced-lane wiring (post-matrix)', () => {
|
||||
});
|
||||
|
||||
describe('evals-periodic.yml sliced-lane wiring', () => {
|
||||
test('the CI job cap covers the live periodic slice census plus setup', () => {
|
||||
type Env = Record<string, string>;
|
||||
const workflow = Bun.YAML.parse(periodicYml) as {
|
||||
env?: Env;
|
||||
jobs: Record<string, {
|
||||
env?: Env;
|
||||
'timeout-minutes': number;
|
||||
strategy?: { matrix: { slice: number[] } };
|
||||
steps: Array<{ run?: string; env?: Env }>;
|
||||
}>;
|
||||
};
|
||||
const planner = workflow.jobs['plan-slices'];
|
||||
const executor = workflow.jobs['eval-slices'];
|
||||
const plannerSteps = planner.steps.filter(step => step.run?.includes('EVALS_TIER=periodic ') && step.run.includes('--emit-plan '));
|
||||
const executorSteps = executor.steps.filter(step => step.run?.includes('--plan '));
|
||||
expect(plannerSteps).toHaveLength(1);
|
||||
expect(executorSteps).toHaveLength(1);
|
||||
const cliArgs = (run: string) => {
|
||||
const command = /\bbun(?: --no-install)? run scripts\/test-paid-shards\.ts /.exec(run);
|
||||
expect(command).not.toBeNull();
|
||||
return run.slice(command!.index + command![0].length)
|
||||
.replace(/\$\{\{\s*matrix\.slice\s*\}\}/g, '1').trim().split(/\s+/);
|
||||
};
|
||||
const plannerEnv = { ...workflow.env, ...planner.env, ...plannerSteps[0].env };
|
||||
const plannerOptions = parseCliOptions(cliArgs(plannerSteps[0].run!), plannerEnv);
|
||||
const executorOptions = parseCliOptions(cliArgs(executorSteps[0].run!), {
|
||||
...workflow.env, ...executor.env, ...executorSteps[0].env,
|
||||
});
|
||||
expect(plannerEnv.EVALS_ALL).toBe('1');
|
||||
expect(plannerOptions.tier).toBe('periodic');
|
||||
expect(executorOptions.tier).toBe('periodic');
|
||||
const slices = executor.strategy!.matrix.slice;
|
||||
expect(slices).toEqual(Array.from({ length: plannerOptions.slices }, (_, i) => i + 1));
|
||||
const manifest = buildRunManifest({
|
||||
tier: plannerOptions.tier, sliceCount: plannerOptions.slices,
|
||||
evalsAll: true, env: plannerEnv, rootDir: ROOT,
|
||||
});
|
||||
// Resolve the same per-file walls and overlay admission limit as execution.
|
||||
const explicitWall = executorOptions.timeoutExplicit ? executorOptions.timeoutMs : undefined;
|
||||
const setupAllowanceMinutes = 20;
|
||||
const allowances = slices.map(slice => {
|
||||
const files = manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice).map(entry => entry.file);
|
||||
const normal = files.filter(file => !isOverlayTestFile(file));
|
||||
const overlay = files.filter(isOverlayTestFile);
|
||||
const bound = (group: string[], jobs: number) => paidShardWallUpperBoundMs(group, jobs, explicitWall);
|
||||
return (bound(normal, executorOptions.jobs) + bound(overlay, Math.min(executorOptions.jobs, OVERLAY_MAX_ACTIVE_SHARDS))) / 60_000;
|
||||
});
|
||||
expect(Math.max(...allowances)).toBeGreaterThan(0);
|
||||
const requiredMinutes = Math.max(...allowances) + setupAllowanceMinutes;
|
||||
expect(executor['timeout-minutes'],
|
||||
`periodic slice allowances ${allowances.join(', ')} minutes + ${setupAllowanceMinutes} minutes setup require ${requiredMinutes} CI minutes`,
|
||||
).toBeGreaterThanOrEqual(requiredMinutes);
|
||||
});
|
||||
const lanes = [
|
||||
{ source: periodicYml, name: 'evals-periodic.yml', executor: 'eval-slices', prefix: 'periodic_', tier: 'periodic' },
|
||||
{ source: periodicYml, name: 'evals-periodic.yml', executor: 'gate-census', prefix: 'gate_', tier: 'gate' },
|
||||
{ source: evalsYml, name: 'evals.yml', executor: 'eval-slices', prefix: '', tier: 'gate' },
|
||||
{ source: marathonYml, name: 'evals-marathon.yml', executor: 'eval-slices', prefix: '', tier: 'marathon' },
|
||||
] as const;
|
||||
|
||||
test('planner/executor/report tier=periodic and slice counts agree', () => {
|
||||
for (const lane of lanes) {
|
||||
test(`${lane.name}:${lane.executor} — the planned CI job cap covers every slice's supervised wall plus setup, and every slice starts at once`, () => {
|
||||
const { emit, execute, executor } = expectPlannedExecutor(lane.source, lane.executor, lane.prefix);
|
||||
const cliArgs = (run: string) => {
|
||||
const command = /\bbun(?: --no-install)? run scripts\/test-paid-shards\.ts /.exec(run);
|
||||
expect(command).not.toBeNull();
|
||||
return run.slice(command!.index + command![0].length)
|
||||
.replace(/\$\{\{\s*matrix\.slice\s*\}\}/g, '1').trim().split(/\s+/);
|
||||
};
|
||||
const workflow = Bun.YAML.parse(lane.source) as { env?: Record<string, string> };
|
||||
// The complete census (EVALS_ALL) is the largest plan any event can produce.
|
||||
const plannerEnv = { ...workflow.env, ...emit.env, EVALS_ALL: '1', EVALS_PROFILE: 'full' };
|
||||
const planned = parseCliOptions(cliArgs(emit.run!), plannerEnv);
|
||||
const active = parseCliOptions(cliArgs(execute.run!), { ...workflow.env, ...executor.env, ...execute.env, EVALS_PROFILE: 'full' });
|
||||
expect(planned.tier).toBe(lane.tier);
|
||||
expect(active.tier).toBe(lane.tier);
|
||||
expect(active.jobs).toBe(planned.jobs);
|
||||
const manifest = buildRunManifest({ tier: planned.tier, profile: 'full', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs,
|
||||
evalsAll: true, env: plannerEnv, rootDir: ROOT, skipJudges: planned.skipJudges });
|
||||
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
|
||||
manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === i + 1)).map(entry => entry.file), planned.jobs));
|
||||
const requiredMinutes = Math.ceil(Math.max(0, ...walls) / 60_000) + CI_SETUP_ALLOWANCE_MINUTES;
|
||||
expect(CI_SETUP_ALLOWANCE_MINUTES).toBe(20);
|
||||
expect(manifest.plan!.ciTimeoutMinutes, `slice walls ${walls.join(', ')}ms`).toBe(requiredMinutes);
|
||||
// GitHub-hosted-style job ceiling: a plan past it must be split, not truncated.
|
||||
expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
|
||||
expect(manifest.sliceCount, `${lane.name}:${lane.executor} plans more slices than max-parallel starts at once`)
|
||||
.toBeLessThanOrEqual(executor.strategy!['max-parallel']);
|
||||
});
|
||||
}
|
||||
|
||||
test('planner/executor/report tier=periodic agree and plan with the ~9-minute budget', () => {
|
||||
expect(periodicYml).toMatch(/EVALS_TIER=periodic bun --no-install run scripts\/test-paid-shards\.ts --tier periodic --emit-plan/);
|
||||
expect(periodicYml).toMatch(/EVALS_TIER=periodic bun run scripts\/test-paid-shards\.ts --tier periodic --plan .* --slice /);
|
||||
expect(periodicYml).toMatch(/EVALS_TIER=periodic bun --no-install run scripts\/test-paid-shards\.ts --tier periodic --report /);
|
||||
const planned = plannedSlices(periodicYml);
|
||||
const matrices = matrixSlices(periodicYml);
|
||||
// Periodic work and the full gate census have distinct immutable plans.
|
||||
expect(planned).toHaveLength(2);
|
||||
expect(matrices).toHaveLength(2);
|
||||
for (const [index, count] of planned.entries()) {
|
||||
expect(matrices[index]).toEqual(Array.from({ length: count }, (_, i) => i + 1));
|
||||
expect(plannerSites(periodicYml)).toEqual([
|
||||
{ manifest: '/tmp/paid-plan/manifest.json', budgetSeconds: 540, jobs: 2 },
|
||||
{ manifest: '/tmp/gate-census-plan/manifest.json', budgetSeconds: 540, jobs: 2 },
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('evals-marathon.yml non-blocking lane', () => {
|
||||
const workflow = Bun.YAML.parse(marathonYml) as { on: Record<string, unknown>; env: Record<string, string>; jobs: Record<string, Job> };
|
||||
|
||||
test('runs weekly and on dispatch, always fresh, with its own fail-closed report and tracking issue', () => {
|
||||
expect(Object.keys(workflow.on).sort()).toEqual(['schedule', 'workflow_dispatch']);
|
||||
expect(workflow.env).toMatchObject({ EVALS_PROFILE: 'full', EVALS_FRESH: '1', EVALS_CACHE_PURPOSE: 'marathon' });
|
||||
expect(marathonYml).not.toContain('actions/cache');
|
||||
expect(marathonYml).toMatch(/EVALS_TIER=marathon bun --no-install run scripts\/test-paid-shards\.ts --tier marathon --emit-plan \/tmp\/marathon-plan\/manifest\.json --slice-budget 1 --jobs 1/);
|
||||
expect(marathonYml).toMatch(/EVALS_TIER=marathon bun run scripts\/test-paid-shards\.ts --tier marathon --plan .* --slice /);
|
||||
const report = workflow.jobs.report!;
|
||||
expect(report.needs).toEqual(['plan-slices', 'eval-slices']);
|
||||
const reconcile = report.steps.find(step => step.id === 'reconcile')!;
|
||||
expect(reconcile.run).toContain('EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --report /tmp/marathon-report');
|
||||
const guards = report.steps.filter(step => /Upsert tracking|Fail the workflow/.test(step.name ?? ''));
|
||||
expect(guards).toHaveLength(2);
|
||||
for (const step of guards) {
|
||||
expect((step as { if?: string }).if).toContain("steps.reconcile.outputs.exit != '0'");
|
||||
expect((step as { if?: string }).if).toContain("needs.eval-slices.result != 'success'");
|
||||
}
|
||||
expect(marathonYml).toContain('Weekly marathon evals: red lane needs triage');
|
||||
});
|
||||
|
||||
test('the blocking lanes never plan or execute the marathon tier', () => {
|
||||
for (const source of [evalsYml, periodicYml]) {
|
||||
expect(source).not.toContain('--tier marathon');
|
||||
expect(source).not.toContain('EVALS_TIER=marathon');
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('shared setup composites (both surviving lanes)', () => {
|
||||
test('both lanes register skills through the shared composite', () => {
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) {
|
||||
describe('shared setup composites (every paid lane)', () => {
|
||||
test('every lane registers skills through the shared composite', () => {
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) {
|
||||
expect(source, `${name} must use the register-gstack-skills composite`)
|
||||
.toContain('uses: ./.github/actions/register-gstack-skills');
|
||||
// No inline re-implementation creeping back beside the composite.
|
||||
@@ -218,7 +256,7 @@ describe('shared setup composites (both surviving lanes)', () => {
|
||||
for (const action of ['seed-claude-config', 'restore-deps', 'fix-bun-temp']) {
|
||||
expect(fs.existsSync(path.join(ROOT, '.github', 'actions', action, 'action.yml')), `missing composite: ${action}`).toBe(true);
|
||||
}
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml]] as const) {
|
||||
for (const [name, source] of [['evals.yml', evalsYml], ['evals-periodic.yml', periodicYml], ['evals-marathon.yml', marathonYml]] as const) {
|
||||
expect(source, `${name} must use seed-claude-config`).toContain('uses: ./.github/actions/seed-claude-config');
|
||||
expect(source, `${name} must use restore-deps`).toContain('uses: ./.github/actions/restore-deps');
|
||||
expect(source, `${name} must use fix-bun-temp`).toContain('uses: ./.github/actions/fix-bun-temp');
|
||||
|
||||
+103
-40
@@ -46,70 +46,133 @@ export const ALL_TIERS = {
|
||||
/** Supervision reserve added to every registered whole-file wall. */
|
||||
export const SHARD_RESERVE_MS = 2 * 60_000;
|
||||
|
||||
/** Whole-file supervision must cover each existing attempt and its retry.
|
||||
* These fixtures already allow 25 minutes per case; the old 30-minute
|
||||
* wall could kill a second attempt after five minutes. No case budget grows.
|
||||
/**
|
||||
* Retry policy (approved 2026-09-29): a timed-out attempt is a verdict. Bun's
|
||||
* --retry reruns a failed case after it may have spent its whole budget, so an
|
||||
* automatic retry is kept only where one more attempt is short: every case of
|
||||
* the file has a per-attempt budget of at most RETRY_MAX_CASE_MS, the CAPTURE
|
||||
* tier plus its recording grace. Those failures are fast flake classes (API
|
||||
* blips, tool hiccups) and a retry costs at most one more short attempt. Files
|
||||
* with any longer case run once. Per-case budgets never change with this rule.
|
||||
*/
|
||||
export const RETRY_MAX_CASE_MS = CAPTURE_MS + 15_000;
|
||||
|
||||
export function retriesWithinCaseCap(caseMs: number, configuredRetries: number): number {
|
||||
return caseMs <= RETRY_MAX_CASE_MS ? configuredRetries : 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Unregistered paid files that keep one automatic retry: every case budget is
|
||||
* JUDGE or CAPTURE tier (test/paid-retry-supervision.test.ts scans each source).
|
||||
* Registered rows below derive retries from their declared caseMs; every other
|
||||
* paid file runs once.
|
||||
*/
|
||||
export const SHORT_CASE_RETRY_FILES: readonly string[] = [
|
||||
'test/codex-e2e-sol-scope.test.ts',
|
||||
'test/llm-judge-recommendation.test.ts',
|
||||
'test/skill-e2e-ask-user-question-format-compliance.test.ts',
|
||||
'test/skill-e2e-benchmark-providers.test.ts',
|
||||
'test/skill-e2e-bws.test.ts',
|
||||
'test/skill-e2e-context-skills.test.ts',
|
||||
'test/skill-e2e-coverage-audit.test.ts',
|
||||
'test/skill-e2e-diagram.test.ts',
|
||||
'test/skill-e2e-first-task-scaffold.test.ts',
|
||||
'test/skill-e2e-gbrain-roundtrip-local.test.ts',
|
||||
'test/skill-e2e-hermetic-canary.test.ts',
|
||||
'test/skill-e2e-investigate-owned-completion.test.ts',
|
||||
'test/skill-e2e-investigate-owned-termination.test.ts',
|
||||
'test/skill-e2e-learnings.test.ts',
|
||||
'test/skill-e2e-plan-tune.test.ts',
|
||||
'test/skill-e2e-qa-functional-fix.test.ts',
|
||||
'test/skill-e2e-qa-functional.test.ts',
|
||||
'test/skill-e2e-review-army.test.ts',
|
||||
'test/skill-e2e-review.test.ts',
|
||||
'test/skill-e2e-session-intelligence.test.ts',
|
||||
'test/skill-e2e-setup-gbrain-bad-token.test.ts',
|
||||
'test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts',
|
||||
'test/skill-e2e-setup-gbrain-remote.test.ts',
|
||||
'test/skill-e2e-ship-hook-consent.test.ts',
|
||||
'test/skill-e2e-ship-hook-refresh.test.ts',
|
||||
'test/skill-e2e-ship-skip.test.ts',
|
||||
'test/skill-e2e-sync-gbrain-readiness.test.ts',
|
||||
'test/skill-e2e-third-party-actions.test.ts',
|
||||
'test/skill-e2e-triage.test.ts',
|
||||
'test/skill-routing-e2e.test.ts',
|
||||
];
|
||||
|
||||
/** Whole-file supervision covers every attempt the retry policy allows.
|
||||
* These fixtures allow 25 minutes per case, so they run once.
|
||||
* Reserve the sequential upper bound even when Bun runs sibling cases together.
|
||||
*/
|
||||
export const FINDING_RETRY_BUDGETS = [
|
||||
{ file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 },
|
||||
].map(({ file, cases }) => ({
|
||||
file, cases,
|
||||
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
testMs: 1_500_000,
|
||||
retries: 1,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: cases * 1_500_000 * 2 + SHARD_RESERVE_MS,
|
||||
}));
|
||||
].map(({ file, cases }) => {
|
||||
const retries = retriesWithinCaseCap(1_500_000, 1);
|
||||
return {
|
||||
file, cases,
|
||||
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
testMs: 1_500_000,
|
||||
caseMs: 1_500_000,
|
||||
retries,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: cases * 1_500_000 * (retries + 1) + SHARD_RESERVE_MS,
|
||||
};
|
||||
});
|
||||
|
||||
/** Three existing captures and one configured retry; only supervision grows. */
|
||||
/** Three existing captures in one 16-minute case, so the file runs once. */
|
||||
export const AUQ_CONSISTENCY_RETRY_BUDGET = {
|
||||
file: 'test/skill-e2e-auq-consistency.test.ts',
|
||||
id: 'auq-consistency-existing-retry-v1',
|
||||
cases: 1,
|
||||
testMs: 3 * CAPTURE_MS + 60_000,
|
||||
retries: 1,
|
||||
caseMs: 3 * CAPTURE_MS + 60_000,
|
||||
retries: retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1),
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: (3 * CAPTURE_MS + 60_000) * 2 + SHARD_RESERVE_MS,
|
||||
shardMs: (3 * CAPTURE_MS + 60_000) * (retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1) + 1) + SHARD_RESERVE_MS,
|
||||
} as const;
|
||||
|
||||
/** These fixtures have a fixed case count in every supported tier. */
|
||||
export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET];
|
||||
|
||||
/** Whole-file walls cover all existing cases and retries, even if Bun runs them
|
||||
* sequentially. Mixed-tier files reserve their larger complete tier, never a
|
||||
* currently selected subset. These rows add no case-count or model-work policy.
|
||||
* The 10-second terms preserve the existing Codex/recording finalization grace.
|
||||
/** Whole-file walls cover all existing cases and every allowed attempt, even if
|
||||
* Bun runs them sequentially. Mixed-tier files reserve their larger complete
|
||||
* tier, never a currently selected subset. caseMs is the longest single case
|
||||
* budget, which decides the retry (RETRY_MAX_CASE_MS). These rows add no
|
||||
* case-count or model-work policy. The 10-second terms preserve the existing
|
||||
* Codex/recording finalization grace.
|
||||
*/
|
||||
export const FILE_RETRY_BUDGETS = [
|
||||
...STRICT_RETRY_CASE_BUDGETS,
|
||||
...[
|
||||
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), retries: 1 },
|
||||
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 5 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 5 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
// Seventeen workflow judges include their 10s recording grace; the other
|
||||
// seven judges retain 120s. Supervise all 24 and the existing one retry.
|
||||
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
|
||||
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, retries: 2 },
|
||||
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
|
||||
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 2 },
|
||||
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
|
||||
// Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s.
|
||||
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), retries: 1 },
|
||||
].map(({ file, attemptMs, retries }) => ({
|
||||
file, attemptMs, retries,
|
||||
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
|
||||
})),
|
||||
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS, configuredRetries: 1 },
|
||||
].map(({ file, attemptMs, caseMs, configuredRetries }) => {
|
||||
const retries = retriesWithinCaseCap(caseMs, configuredRetries);
|
||||
return {
|
||||
file, attemptMs, caseMs, retries,
|
||||
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
|
||||
};
|
||||
}),
|
||||
];
|
||||
|
||||
/** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */
|
||||
|
||||
@@ -22,17 +22,18 @@ const fakeEnv = {
|
||||
|
||||
describe('overlay file policy', () => {
|
||||
test('grouped planning isolates every overlay and preserves ordinary retries', () => {
|
||||
const workflow = 'test/skill-e2e-workflow.test.ts';
|
||||
const files = [...overlayFiles, normalFile, workflow];
|
||||
// Two short-case files keep their one retry (timeout-is-a-verdict rule).
|
||||
const workflow = 'test/skill-e2e-review.test.ts';
|
||||
const files = [...overlayFiles, 'test/skill-e2e-triage.test.ts', workflow];
|
||||
for (const maxFilesPerShard of [2, 3, 10]) {
|
||||
const shards = planPaidShards(files, { maxFilesPerShard });
|
||||
expect(shards.flat().sort()).toEqual([...files].sort());
|
||||
for (const file of overlayFiles) expect(shards).toContainEqual([file]);
|
||||
const workflowShard = shards.find(shard => shard.includes(workflow))!;
|
||||
expect(workflowShard.some(isOverlayTestFile)).toBe(false);
|
||||
expect(retriesForFiles(workflowShard)).toBe(2);
|
||||
expect(retriesForFiles(workflowShard)).toBe(1);
|
||||
const args = buildPaidShardArgs(workflowShard, resolvePaidShardTimeoutMs(workflowShard), 2, retriesForFiles(workflowShard));
|
||||
expect(args[args.indexOf('--retry') + 1]).toBe('2');
|
||||
expect(args[args.indexOf('--retry') + 1]).toBe('1');
|
||||
expect(planPaidShards(files.map(file => file.replaceAll('/', '\\')), { maxFilesPerShard })).toEqual(shards);
|
||||
}
|
||||
});
|
||||
@@ -65,8 +66,10 @@ describe('overlay file policy', () => {
|
||||
for (const file of [normalFile, 'test/skill-e2e-overlay-harness.test.ts', 'test/model-overlays.test.ts']) {
|
||||
expect(isOverlayTestFile(file)).toBe(false);
|
||||
expect(resolvePaidShardTimeoutMs([file])).toBe(DEFAULT_SHARD_TIMEOUT_MS);
|
||||
expect(retriesForFiles([file])).toBe(1);
|
||||
// Not overlays; unlisted files run once because their case budget is unknown.
|
||||
expect(retriesForFiles([file])).toBe(0);
|
||||
}
|
||||
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1);
|
||||
expect(resolvePaidShardTimeoutMs([normalFile], 1234)).toBe(1234);
|
||||
expect(resolvePaidShardTimeoutMs([overlayFiles[0]], 1_900_000)).toBe(1_900_000);
|
||||
expect(() => resolvePaidShardTimeoutMs([overlayFiles[0]], 1_800_000)).toThrow('explicit wall');
|
||||
@@ -157,8 +160,8 @@ describe('overlay manifest affinity and CI capacity', () => {
|
||||
const workflow = Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows/evals-periodic.yml'), 'utf8')) as {
|
||||
jobs: Record<string, {
|
||||
steps: Array<{ run?: string; env?: NodeJS.ProcessEnv }>;
|
||||
strategy: { matrix: { slice: number[] } };
|
||||
'timeout-minutes': number;
|
||||
strategy: { matrix: { slice: string } };
|
||||
'timeout-minutes': string;
|
||||
}>;
|
||||
};
|
||||
const job = workflow.jobs['eval-slices'];
|
||||
@@ -166,13 +169,20 @@ describe('overlay manifest affinity and CI capacity', () => {
|
||||
const jobs = parseCliOptions([], step.env).jobs;
|
||||
expect(jobs).toBe(2);
|
||||
expect(parseCliOptions([], step.env).withinShardConcurrency).toBe(2);
|
||||
expect(job.strategy.matrix.slice).toEqual([1, 2, 3, 4, 5, 6, 7]);
|
||||
expect(job.strategy.matrix.slice).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}');
|
||||
expect(job['timeout-minutes']).toBe('${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}');
|
||||
const normalMinutes = Math.ceil(18 / jobs) * resolvePaidShardTimeoutMs([normalFiles[0]]) / 60_000;
|
||||
const overlayMinutes = Math.ceil(overlayFiles.length / OVERLAY_MAX_ACTIVE_SHARDS)
|
||||
* Math.max(...overlayFiles.map(file => resolvePaidShardTimeoutMs([file]))) / 60_000;
|
||||
expect(normalMinutes).toBe(270);
|
||||
expect(overlayMinutes).toBe(122);
|
||||
expect(job['timeout-minutes']).toBeGreaterThanOrEqual(Math.max(normalMinutes, overlayMinutes) + 20);
|
||||
// The CI budget plan (what the workflow runs) keeps the four overlays
|
||||
// in one final one-at-a-time slice and its job cap covers them.
|
||||
const budget = buildRunManifest({ ...opts, sliceCount: undefined, sliceBudgetMs: 540_000, jobs });
|
||||
const lastSlice = budget.entries.filter(e => e.slice === budget.sliceCount).map(e => e.file).sort();
|
||||
expect(lastSlice).toEqual([...overlayFiles].sort());
|
||||
expect(budget.plan!.ciTimeoutMinutes).toBeGreaterThanOrEqual(overlayMinutes + 20);
|
||||
expect(budget.plan!.ciTimeoutMinutes).toBe(Math.ceil(overlayMinutes) + 20);
|
||||
|
||||
// Gate selection keeps its original periodic exclusion and all six
|
||||
// ordinary slices; reservation does not spend an empty slot in gate.
|
||||
|
||||
@@ -150,7 +150,8 @@ describe('PR profile paid-runner integration', () => {
|
||||
expect(() => parseRunManifest(JSON.stringify(injected))).toThrow('outside its PR case selection');
|
||||
for (const action of ['remove', 'skip', 'duplicate'] as const) {
|
||||
const missing = structuredClone(manifest);
|
||||
const file = 'test/skill-e2e-plan.test.ts';
|
||||
// plan.test is case-sharded: its PR case runs as `<file>#<case id>`.
|
||||
const file = manifest.entries.find(entry => entry.status === 'planned' && entry.file.startsWith('test/skill-e2e-plan.test.ts#'))!.file;
|
||||
if (action === 'remove') missing.entries = missing.entries.filter(entry => entry.file !== file);
|
||||
if (action === 'skip') missing.entries.find(entry => entry.file === file)!.status = 'skipped-by-diff';
|
||||
if (action === 'duplicate') missing.entries.push({ ...missing.entries.find(entry => entry.file === file)! });
|
||||
|
||||
@@ -4,34 +4,69 @@ import { join } from 'node:path';
|
||||
import {
|
||||
buildPaidShardArgs, buildRunManifest, parseRunManifest, planPaidShards,
|
||||
DEFAULT_JOBS, parseCliOptions, paidShardWallUpperBoundMs, resolvePaidShardBudget, retriesForFiles, verifySliceResults, collectPaidTestFiles, selectPaidTestFiles,
|
||||
shardFile, sliceExecutionOrder, sliceSupervisedWallMs, CASE_SHARDED_FILES,
|
||||
} from '../scripts/test-paid-shards';
|
||||
import {
|
||||
ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS,
|
||||
ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, RETRY_MAX_CASE_MS, SHORT_CASE_RETRY_FILES,
|
||||
FINDING_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS,
|
||||
} from './helpers/eval-budgets';
|
||||
|
||||
import { E2E_TOUCHFILES } from './helpers/touchfiles';
|
||||
|
||||
const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8');
|
||||
const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file));
|
||||
// Walls cover every attempt the retry rule allows: files with a case budget
|
||||
// past RETRY_MAX_CASE_MS run once (a timed-out attempt is a verdict).
|
||||
const expectedWalls = {
|
||||
'test/skill-e2e-qa-callers.test.ts': 3_270_000,
|
||||
'test/skill-e2e-shared-libs-paths.test.ts': 3_720_000,
|
||||
'test/skill-e2e-ship-docsync.test.ts': 10_920_000,
|
||||
'test/skill-e2e-shared-libs-paths.test.ts': 1_920_000,
|
||||
'test/skill-e2e-ship-docsync.test.ts': 5_520_000,
|
||||
'test/skill-llm-eval.test.ts': 6_220_000,
|
||||
'test/skill-e2e-auq-consistency.test.ts': 2_040_000,
|
||||
'test/skill-e2e-auq-consistency.test.ts': 1_080_000,
|
||||
'test/skill-e2e-auq-matrix.test.ts': 3_720_000,
|
||||
'test/skill-e2e-plan-format.test.ts': 2_600_000,
|
||||
'test/skill-e2e-auto-decide-preserved.test.ts': 1_920_000,
|
||||
'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_920_000,
|
||||
'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_920_000,
|
||||
'test/skill-e2e-plan-design-finding-floor.test.ts': 1_920_000,
|
||||
'test/skill-e2e-plan-devex-finding-floor.test.ts': 1_920_000,
|
||||
'test/skill-e2e-plan-mode-no-op.test.ts': 9_120_000,
|
||||
'test/skill-e2e-plan-ceo-mode-routing.test.ts': 2_520_000,
|
||||
'test/skill-e2e-plan-eng-plan-mode.test.ts': 2_520_000,
|
||||
'test/skill-e2e-auto-decide-preserved.test.ts': 1_020_000,
|
||||
'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_020_000,
|
||||
'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_020_000,
|
||||
'test/skill-e2e-plan-design-finding-floor.test.ts': 1_020_000,
|
||||
'test/skill-e2e-plan-devex-finding-floor.test.ts': 1_020_000,
|
||||
'test/skill-e2e-plan-mode-no-op.test.ts': 3_120_000,
|
||||
'test/skill-e2e-plan-ceo-mode-routing.test.ts': 1_320_000,
|
||||
'test/skill-e2e-plan-eng-plan-mode.test.ts': 1_320_000,
|
||||
'test/skill-e2e-plan-prosons.test.ts': 2_600_000,
|
||||
'test/skill-e2e-plan.test.ts': 7_320_000,
|
||||
'test/skill-e2e-plan.test.ts': 3_720_000,
|
||||
};
|
||||
|
||||
test('retry rule: only files whose every case is CAPTURE tier or shorter retry; longer cases run once', () => {
|
||||
expect(RETRY_MAX_CASE_MS).toBe(ALL_TIERS.CAPTURE_MS + 15_000);
|
||||
for (const row of FILE_RETRY_BUDGETS) {
|
||||
expect(row.retries, row.file).toBe(row.caseMs <= RETRY_MAX_CASE_MS ? (row.file.endsWith('plan-mode-no-op.test.ts') ? 2 : 1) : 0);
|
||||
expect(retriesForFiles([row.file])).toBe(row.retries);
|
||||
}
|
||||
expect(FILE_RETRY_BUDGETS.filter(row => row.retries > 0).map(row => row.file).sort()).toEqual([
|
||||
'test/skill-e2e-auq-matrix.test.ts', 'test/skill-e2e-plan-format.test.ts', 'test/skill-e2e-plan-prosons.test.ts',
|
||||
'test/skill-e2e-qa-callers.test.ts', 'test/skill-llm-eval.test.ts',
|
||||
]);
|
||||
const paid = collectPaidTestFiles();
|
||||
for (const file of SHORT_CASE_RETRY_FILES) {
|
||||
expect(paid, `stale SHORT_CASE_RETRY_FILES entry: ${file}`).toContain(file);
|
||||
expect(FILE_RETRY_BUDGETS.some(row => row.file === file)).toBe(false);
|
||||
const source = read(file);
|
||||
// Declared short budgets only: a JUDGE/CAPTURE tier or a literal at most the
|
||||
// cap, no longer tier and no ms literal past the cap.
|
||||
const literals = [...source.matchAll(/(?<![\w.])(\d{1,3}(?:_\d{3})+|\d{5,})(?![\w.])/g)]
|
||||
.map(match => Number(match[1]!.replace(/_/g, '')));
|
||||
expect(/\b(?:JUDGE_MS|CAPTURE_MS)\b/.test(source) || literals.some(ms => ms >= 60_000 && ms <= RETRY_MAX_CASE_MS), file).toBe(true);
|
||||
expect(source, file).not.toMatch(/\b(?:CAPTURE_LONG_MS|PTY_MS|PTY_LONG_MS|OVERLAY_CASE_[A-Z_]+)\b/);
|
||||
expect(literals.filter(ms => ms > RETRY_MAX_CASE_MS && ms < 10_000_000), file).toEqual([]);
|
||||
expect(retriesForFiles([file])).toBe(1);
|
||||
}
|
||||
for (const file of paid.filter(file => !SHORT_CASE_RETRY_FILES.includes(file) && !FILE_RETRY_BUDGETS.some(row => row.file === file))) {
|
||||
expect(retriesForFiles([file]), file).toBe(0);
|
||||
}
|
||||
expect(retriesForFiles([SHORT_CASE_RETRY_FILES[0]!, 'test/skill-e2e-plan.test.ts'])).toBe(0);
|
||||
});
|
||||
|
||||
test('registration covers exactly the seventeen demonstrated full-file retry gaps', () => {
|
||||
expect(Object.fromEntries(newBudgets.map(row => [row.file, row.shardMs]))).toEqual(expectedWalls);
|
||||
expect(new Set(FILE_RETRY_BUDGETS.map(row => row.file)).size).toBe(19);
|
||||
@@ -86,12 +121,17 @@ test('source allowances retain all captures, cases, and finalization grace', ()
|
||||
]);
|
||||
});
|
||||
|
||||
// Case-sharded files plan one registered case per shard (`<file>#<case id>`).
|
||||
const plannedKey = (file: string) => CASE_SHARDED_FILES.includes(file)
|
||||
? `${file}#${Object.keys(E2E_TOUCHFILES).find(id => E2E_TOUCHFILES[id]!.includes(file))}` : file;
|
||||
|
||||
for (const row of newBudgets) {
|
||||
const key = plannedKey(row.file);
|
||||
const manifest = () => ({ version: 1 as const, tier: 'periodic' as const, evalsAll: false,
|
||||
sliceCount: 1, entries: [{ file: row.file, slice: 1, status: 'planned' as const, budget: resolvePaidShardBudget([row.file]) }] });
|
||||
sliceCount: 1, entries: [{ file: key, slice: 1, status: 'planned' as const, budget: resolvePaidShardBudget([key]) }] });
|
||||
const result = (count = 1) => [{ version: 1 as const, tier: 'periodic' as const, sliceIndex: 1, sliceCount: 1,
|
||||
outcomes: [{ files: [row.file], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count,
|
||||
skippedTests: 0, budget: resolvePaidShardBudget([row.file]) }] }];
|
||||
outcomes: [{ files: [key], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count,
|
||||
skippedTests: 0, budget: resolvePaidShardBudget([key]) }] }];
|
||||
|
||||
test(`${row.file}: full wall and existing retries propagate through planning`, () => {
|
||||
expect(retriesForFiles([row.file])).toBe(row.retries);
|
||||
@@ -132,8 +172,11 @@ test('fixed AUQ count remains strict while mixed-tier files keep ordinary case h
|
||||
[AUQ_CONSISTENCY_RETRY_BUDGET.file, 0, false],
|
||||
[AUQ_CONSISTENCY_RETRY_BUDGET.file, 1, true],
|
||||
[AUQ_CONSISTENCY_RETRY_BUDGET.file, 2, false],
|
||||
['test/skill-e2e-plan.test.ts', 5, true],
|
||||
['test/skill-e2e-plan.test.ts', 7, true],
|
||||
['test/skill-e2e-ship-docsync.test.ts', 5, true],
|
||||
['test/skill-e2e-ship-docsync.test.ts', 7, true],
|
||||
// A case shard of a case-sharded registered file executes exactly its case.
|
||||
[plannedKey('test/skill-e2e-plan.test.ts'), 1, true],
|
||||
[plannedKey('test/skill-e2e-plan.test.ts'), 2, false],
|
||||
] as const) {
|
||||
const budget = resolvePaidShardBudget([file]);
|
||||
const manifest: any = { version: 1, tier: 'periodic', evalsAll: true, sliceCount: 1,
|
||||
@@ -158,7 +201,7 @@ test('quality judge supervision includes the added judge without changing ordina
|
||||
expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS');
|
||||
expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000);
|
||||
expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([
|
||||
...Array(2).fill([1, 1500000, 1, 3120000]),
|
||||
...Array(2).fill([1, 1500000, 0, 1620000]),
|
||||
]);
|
||||
for (const tier of ['gate', 'periodic'] as const) {
|
||||
const m = buildRunManifest({ tier, sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
@@ -184,7 +227,7 @@ test('detached PR fallback and release commands cover their actual default worke
|
||||
const prFloor = Math.ceil((Math.ceil(fullGateFiles.length / prWorkers) * 1_800_000 + fullGateFiles.reduce(
|
||||
(total, file) => total + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - 1_800_000), 0,
|
||||
)) / 1000 * 1.05);
|
||||
expect(prFloor).toBe(74_981);
|
||||
expect(prFloor).toBe(71_957);
|
||||
expect(prWall).toBe(92_820_000);
|
||||
expect(prWall).toBeGreaterThanOrEqual(paidShardWallUpperBoundMs(files, prWorkers) + 120_000);
|
||||
|
||||
@@ -203,8 +246,8 @@ test('detached PR fallback and release commands cover their actual default worke
|
||||
)) / 1000 * 1.05));
|
||||
}
|
||||
const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000;
|
||||
expect(releaseFloors).toEqual([42_851, 37_727]);
|
||||
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(80_578);
|
||||
expect(releaseFloors).toEqual([26_597, 30_797]);
|
||||
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(57_394);
|
||||
expect(detachedReleaseWall).toBe(116_700_000);
|
||||
expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000);
|
||||
});
|
||||
@@ -217,10 +260,10 @@ const cliOptions = (step: { run: string; env?: Record<string, string> }) => {
|
||||
return parseCliOptions(args, step.env ?? {});
|
||||
};
|
||||
|
||||
test('both gate executors cover the complete census without increasing aggregate workers', () => {
|
||||
test('both gate executors plan the complete census and supervise every planned slice', () => {
|
||||
const periodic: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml'));
|
||||
const main: any = Bun.YAML.parse(read('.github/workflows/evals.yml'));
|
||||
for (const [workflow, jobName, workers, slices] of [[main, 'eval-slices', 2, 7], [periodic, 'gate-census', 1, 7]] as const) {
|
||||
for (const [workflow, jobName, prefix, skipJudges] of [[main, 'eval-slices', '', false], [periodic, 'gate-census', 'gate_', true]] as const) {
|
||||
const planner = workflow.jobs['plan-slices'];
|
||||
const executor = workflow.jobs[jobName];
|
||||
const emit = planner.steps.filter((step: any) => step.run?.includes('EVALS_TIER=gate ') && step.run.includes('--emit-plan '));
|
||||
@@ -230,41 +273,35 @@ test('both gate executors cover the complete census without increasing aggregate
|
||||
const planned = cliOptions(emit[0]), active = cliOptions(execute[0]);
|
||||
expect(planned.tier).toBe('gate');
|
||||
expect(active.tier).toBe('gate');
|
||||
expect(active.jobs).toBe(workers);
|
||||
expect(active.jobs).toBe(2);
|
||||
expect(planned.jobs).toBe(active.jobs);
|
||||
expect(planned.sliceBudgetMs).toBe(540_000);
|
||||
expect(planned.skipJudges).toBe(skipJudges);
|
||||
expect(execute[0].env.EVALS_CONCURRENCY).toBe('2');
|
||||
expect(executor.strategy['fail-fast']).toBe(false);
|
||||
expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: slices }, (_, i) => i + 1));
|
||||
expect(planned.slices).toBe(slices);
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceCount: planned.slices, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(46);
|
||||
const files = manifest.entries.filter(row => row.status === 'planned').map(row => row.file);
|
||||
expect(new Set(files).size).toBe(46);
|
||||
expect(executor.strategy.matrix.slice).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}slices) }}`);
|
||||
expect(executor['timeout-minutes']).toBe(`\${{ fromJSON(needs.plan-slices.outputs.${prefix}timeout_minutes) }}`);
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs, evalsAll: true, env: { EVALS_ALL: '1' }, skipJudges });
|
||||
const files = [...new Set(manifest.entries.filter(row => row.status === 'planned').map(row => shardFile(row.file)))];
|
||||
expect(files).toContain('test/skill-e2e-ship-skip.test.ts');
|
||||
expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected.sort());
|
||||
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
|
||||
manifest.entries.filter(row => row.status === 'planned' && row.slice === slice).map(row => row.file), workers,
|
||||
));
|
||||
expect(executor['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
|
||||
expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected
|
||||
.filter(file => !skipJudges || !file.startsWith('test/skill-llm-eval')).sort());
|
||||
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
|
||||
manifest.entries.filter(row => row.status === 'planned' && row.slice === i + 1)).map(row => row.file), active.jobs));
|
||||
expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
|
||||
expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
|
||||
expect(manifest.sliceCount).toBeLessThanOrEqual(executor.strategy['max-parallel']);
|
||||
if (jobName === 'gate-census') {
|
||||
expect(Math.max(...walls)).toBe(16_440_000);
|
||||
expect(executor['timeout-minutes']).toBe(352);
|
||||
expect(emit[0].env.EVALS_ALL).toBe('1');
|
||||
expect(executor.strategy['max-parallel']).toBe(4);
|
||||
expect(executor.strategy['max-parallel'] * active.jobs).toBe(4);
|
||||
expect(emit[0].run).toContain('--emit-plan /tmp/gate-census-plan/manifest.json');
|
||||
expect(execute[0].run).toContain('--plan /tmp/gate-census-plan/manifest.json --slice ${{ matrix.slice }}');
|
||||
expect(executor.steps.some((step: any) => step.with?.name === 'gate-census-plan')).toBe(true);
|
||||
expect(executor.steps.filter((step: any) => step.run?.includes('--emit-plan '))).toHaveLength(0);
|
||||
} else {
|
||||
expect(Math.max(...walls)).toBe(12_720_000);
|
||||
expect(executor['timeout-minutes']).toBe(265);
|
||||
expect(executor.strategy['max-parallel']).toBe(6);
|
||||
expect(executor.strategy['max-parallel'] * active.jobs).toBe(12);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test('the periodic executor supervises every actual case and retry within its CI wall', () => {
|
||||
test('the periodic executor supervises every actual case and retry within its planned CI wall', () => {
|
||||
const workflow: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml'));
|
||||
const executor = workflow.jobs['eval-slices'];
|
||||
const emit = workflow.jobs['plan-slices'].steps.filter((step: any) =>
|
||||
@@ -274,21 +311,19 @@ test('the periodic executor supervises every actual case and retry within its CI
|
||||
expect(execute).toHaveLength(1);
|
||||
const planned = cliOptions(emit[0]), active = cliOptions(execute[0]);
|
||||
expect(planned.tier).toBe('periodic');
|
||||
expect(planned.slices).toBe(7);
|
||||
expect(planned.sliceBudgetMs).toBe(540_000);
|
||||
expect(active.jobs).toBe(2);
|
||||
expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: planned.slices }, (_, i) => i + 1));
|
||||
const manifest = buildRunManifest({ tier: 'periodic', sliceCount: planned.slices,
|
||||
expect(planned.jobs).toBe(active.jobs);
|
||||
const manifest = buildRunManifest({ tier: 'periodic', sliceBudgetMs: planned.sliceBudgetMs!, jobs: planned.jobs,
|
||||
evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
const census = manifest.entries.filter(row => row.status === 'planned');
|
||||
expect(census).toHaveLength(70);
|
||||
expect(new Set(census.map(row => shardFile(row.file)))).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected));
|
||||
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_220_000);
|
||||
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
|
||||
census.filter(row => row.slice === slice).map(row => row.file), active.jobs,
|
||||
));
|
||||
expect(Math.max(...walls)).toBe(14_680_000);
|
||||
expect(executor.strategy['max-parallel']).toBe(8);
|
||||
expect(executor['timeout-minutes']).toBe(360);
|
||||
expect(executor['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
|
||||
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
|
||||
census.filter(row => row.slice === i + 1)).map(row => row.file), active.jobs));
|
||||
expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
|
||||
expect(manifest.plan!.ciTimeoutMinutes).toBeLessThanOrEqual(360);
|
||||
expect(manifest.sliceCount).toBeLessThanOrEqual(executor.strategy['max-parallel']);
|
||||
});
|
||||
|
||||
test('gate census requires all seven distinct slice results and its own reconciliation', () => {
|
||||
|
||||
+110
-21
@@ -6,8 +6,9 @@
|
||||
* - per-slice selector divergence → ONE planner manifest, executors consume
|
||||
* - hollow lanes → a slice with no artifact is a FAILURE, not an absence
|
||||
* - hollow shards → EVALS_ALL + exit 0 + zero executed tests ≠ pass
|
||||
* - retry parity → the old matrix rows' earned `retries: 2` survive as a
|
||||
* literals map, not folklore
|
||||
* - retry policy → a timed-out attempt is a verdict; only short-case files retry
|
||||
* - budget packing → recorded work packs into ~9-minute executors whose count
|
||||
* and CI job timeout come from the plan
|
||||
*/
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
@@ -21,13 +22,18 @@ import {
|
||||
buildRunManifest,
|
||||
loadPaidTestDurations,
|
||||
mergePaidTestDurations,
|
||||
packBySliceBudget,
|
||||
estimatedSliceMs,
|
||||
sliceExecutionOrder,
|
||||
sliceSupervisedWallMs,
|
||||
writePaidTestDurations,
|
||||
CI_SETUP_ALLOWANCE_MINUTES,
|
||||
paidShardWallUpperBoundMs,
|
||||
parseCliOptions,
|
||||
parseRunManifest,
|
||||
resolvePaidShardBudget,
|
||||
SUPERVISED_WORKER_COUNTS,
|
||||
retriesForFiles,
|
||||
RETRY_OVERRIDES,
|
||||
summarize,
|
||||
summaryExitCode,
|
||||
verifySliceResults,
|
||||
@@ -46,6 +52,7 @@ const outcome = (over: Partial<ShardOutcome>): ShardOutcome => ({
|
||||
elapsedMs: 1000,
|
||||
groupPid: null,
|
||||
executedTests: 3,
|
||||
skippedTests: null,
|
||||
...over,
|
||||
});
|
||||
|
||||
@@ -142,7 +149,8 @@ describe('recorded-duration slice packing', () => {
|
||||
|
||||
test('the PR-profile file set spreads recorded time instead of stacking it', () => {
|
||||
const env = { EVALS_ALL: '1' };
|
||||
const discovered = Object.keys(recorded);
|
||||
// Seed files only: case-shard keys (`<file>#<case>`) come from expansion.
|
||||
const discovered = Object.keys(recorded).filter(key => !key.includes('#'));
|
||||
const load = (files: string[]) => files.reduce((sum, file) => sum + recorded[file], 0);
|
||||
const packed = lanes(buildRunManifest({ tier: 'gate', sliceCount: 6, evalsAll: true, env, discovered })).map(load);
|
||||
const baseline = lanes(buildRunManifest({ tier: 'gate', sliceCount: 6, evalsAll: true, env, discovered, durations: {} })).map(load);
|
||||
@@ -171,6 +179,88 @@ describe('recorded-duration slice packing', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('budget slice packing', () => {
|
||||
const s = (seconds: number) => seconds * 1000;
|
||||
|
||||
test('best-fit packs recorded work under the budget; unknown and over-budget work gets its own runner', () => {
|
||||
const recorded = { 'test/a.test.ts': s(500), 'test/b.test.ts': s(300), 'test/c.test.ts': s(240), 'test/d.test.ts': s(100), 'test/long.test.ts': s(900) };
|
||||
const files = [...Object.keys(recorded), 'test/unknown.test.ts'];
|
||||
const plan = packBySliceBudget(files, s(540), 2, recorded);
|
||||
expect(plan.slices.flat().sort()).toEqual([...files].sort());
|
||||
expect(plan.estimates['test/unknown.test.ts']).toBe(s(540));
|
||||
for (const [index, slice] of plan.slices.entries()) {
|
||||
expect(plan.estimatedSliceMs[index]).toBe(estimatedSliceMs(slice, file => plan.estimates[file]!, 2));
|
||||
if (slice.length > 1) expect(plan.estimatedSliceMs[index]).toBeLessThanOrEqual(s(540));
|
||||
}
|
||||
expect(plan.slices).toContainEqual(['test/long.test.ts']);
|
||||
expect(plan.slices.find(slice => slice.includes('test/unknown.test.ts'))!.length).toBeLessThanOrEqual(2);
|
||||
// Deterministic regardless of discovery order.
|
||||
expect(packBySliceBudget([...files].reverse(), s(540), 2, recorded)).toEqual(plan);
|
||||
const worst = Math.max(...plan.slices.map(slice => sliceSupervisedWallMs(slice, 2)));
|
||||
expect(plan.ciTimeoutMinutes).toBe(Math.ceil(worst / 60_000) + CI_SETUP_ALLOWANCE_MINUTES);
|
||||
expect(packBySliceBudget([], s(540), 2, recorded)).toMatchObject({ slices: [[]], ciTimeoutMinutes: CI_SETUP_ALLOWANCE_MINUTES });
|
||||
});
|
||||
|
||||
test('overlays keep one final slice at their one-at-a-time admission', () => {
|
||||
const overlays = ['test/skill-e2e-overlay-harness-a.test.ts', 'test/skill-e2e-overlay-harness-b.test.ts'];
|
||||
const plan = packBySliceBudget(['test/a.test.ts', ...overlays], s(540), 2, { [overlays[0]!]: s(100), [overlays[1]!]: s(200), 'test/a.test.ts': s(10) });
|
||||
expect(plan.slices.at(-1)).toEqual([overlays[1], overlays[0]]);
|
||||
expect(plan.estimatedSliceMs.at(-1)).toBe(s(300));
|
||||
});
|
||||
|
||||
test('live census plans: multi-file slices stay within the budget and the executor runs longest first', () => {
|
||||
for (const tier of ['gate', 'periodic'] as const) {
|
||||
const manifest = buildRunManifest({ tier, sliceBudgetMs: s(540), jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest);
|
||||
for (let slice = 1; slice <= manifest.sliceCount; slice++) {
|
||||
const entries = sliceExecutionOrder(manifest.entries.filter(entry => entry.status === 'planned' && entry.slice === slice));
|
||||
const estimate = manifest.plan!.estimatedSliceMs[slice - 1]!;
|
||||
if (entries.length > 1 && !entries.some(entry => entry.file.includes('overlay-harness'))) expect(estimate, `${tier} slice ${slice}`).toBeLessThanOrEqual(s(540));
|
||||
expect(entries.map(entry => entry.estimatedMs)).toEqual([...entries.map(entry => entry.estimatedMs!)].sort((a, b) => b - a));
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test('plans fail closed on malformed metadata, mixed modes, and a worker count the plan did not supervise', () => {
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: s(540), jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
for (const broken of [
|
||||
{ ...manifest, plan: { ...manifest.plan!, estimatedSliceMs: [] } },
|
||||
{ ...manifest, plan: { ...manifest.plan!, jobs: 0 } },
|
||||
{ ...manifest, entries: manifest.entries.map(entry => ({ ...entry, estimatedMs: undefined })) },
|
||||
]) expect(() => parseRunManifest(JSON.stringify(broken))).toThrow('slice plan malformed');
|
||||
expect(() => buildRunManifest({ tier: 'gate', sliceCount: 2, sliceBudgetMs: s(540), jobs: 2, evalsAll: true })).toThrow('exactly one');
|
||||
expect(() => buildRunManifest({ tier: 'gate', sliceBudgetMs: s(540), evalsAll: true })).toThrow('explicit positive --jobs');
|
||||
expect(() => parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540'], {})).toThrow('explicit --jobs');
|
||||
expect(() => parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540', '--jobs', '2', '--slices', '3'], {})).toThrow('exactly one');
|
||||
expect(parseCliOptions(['--emit-plan', 'x', '--slice-budget', '540'], { EVALS_JOBS: '2' }).sliceBudgetMs).toBe(s(540));
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-plan-jobs-'));
|
||||
try {
|
||||
const planPath = path.join(dir, 'manifest.json');
|
||||
fs.writeFileSync(planPath, JSON.stringify(manifest));
|
||||
const result = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--plan', planPath, '--slice', '1', '--list', '--jobs', '1'],
|
||||
{ cwd: ROOT, encoding: 'utf8', timeout: 10_000, env: { PATH: path.dirname(process.execPath), HOME: dir, EVALS_TIER: 'gate' } });
|
||||
expect(result.status).toBe(1);
|
||||
expect(result.stderr).toContain('manifest was packed for 2 worker(s) per slice');
|
||||
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
||||
});
|
||||
|
||||
test('the duration seed is per tier and a report rewrite keeps the other tiers', () => {
|
||||
const gate = loadPaidTestDurations(ROOT, 'gate');
|
||||
const periodic = loadPaidTestDurations(ROOT, 'periodic');
|
||||
expect(gate['test/skill-e2e-plan.test.ts']).not.toBe(periodic['test/skill-e2e-plan.test.ts']);
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-durations-'));
|
||||
try {
|
||||
fs.mkdirSync(path.join(dir, 'scripts'));
|
||||
writePaidTestDurations('periodic', { 'test/p.test.ts': 2_000 }, dir);
|
||||
writePaidTestDurations('gate', { 'test/g.test.ts': 3_000 }, dir);
|
||||
expect(loadPaidTestDurations(dir, 'gate')).toEqual({ 'test/g.test.ts': 3_000 });
|
||||
expect(loadPaidTestDurations(dir, 'periodic')).toEqual({ 'test/p.test.ts': 2_000 });
|
||||
fs.writeFileSync(path.join(dir, 'scripts/paid-test-durations.json'), JSON.stringify({ version: 1, durations: { 'test/g.test.ts': 1 } }));
|
||||
expect(loadPaidTestDurations(dir, 'gate')).toEqual({});
|
||||
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
||||
});
|
||||
});
|
||||
|
||||
describe('manifest executor scope', () => {
|
||||
test('list-only validates and prints the selected manifest slice without launching tests or writing results', () => {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-manifest-list-'));
|
||||
@@ -184,8 +274,8 @@ test('local launch sentinel', () => writeFileSync(${JSON.stringify(receipt)}, 't
|
||||
version: 1, tier: 'gate', evalsAll: true, sliceCount: 3, selectionReason: 'local list-only fixture',
|
||||
entries: [
|
||||
{ file, slice: 1, status: 'planned' },
|
||||
{ file: 'test/skill-e2e-plan.test.ts', slice: 2, status: 'planned',
|
||||
budget: resolvePaidShardBudget(['test/skill-e2e-plan.test.ts']) },
|
||||
{ file: 'test/skill-e2e-plan.test.ts#plan-ceo-review', slice: 2, status: 'planned',
|
||||
budget: resolvePaidShardBudget(['test/skill-e2e-plan.test.ts#plan-ceo-review']) },
|
||||
],
|
||||
};
|
||||
const manifestPath = path.join(dir, 'manifest.json');
|
||||
@@ -327,7 +417,7 @@ describe('slice-result reconciliation (report)', () => {
|
||||
tier: 'gate',
|
||||
sliceIndex: index,
|
||||
sliceCount: 2,
|
||||
outcomes: files.map((f) => ({ files: [f], status, exitCode: 0, elapsedMs: 5, executedTests: 2 })),
|
||||
outcomes: files.map((f) => ({ files: [f], status, exitCode: 0, elapsedMs: 5, executedTests: 2, skippedTests: 0 })),
|
||||
});
|
||||
|
||||
test('all slices present and passing → ok', () => {
|
||||
@@ -397,25 +487,24 @@ describe('hollow-shard guard', () => {
|
||||
});
|
||||
|
||||
describe('retry parity', () => {
|
||||
test('registered native workflows preserve main retry policy while overlay attempts stay isolated', () => {
|
||||
test('registered native workflows follow the retry rule while overlay attempts stay isolated', () => {
|
||||
// A 25-minute case is past RETRY_MAX_CASE_MS: its timed-out attempt is the verdict.
|
||||
const native = 'test/skill-e2e-plan-ceo-split-overflow.test.ts';
|
||||
expect(retriesForFiles([native])).toBe(1);
|
||||
expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(1);
|
||||
expect(buildPaidShardArgs([native], 1_800_000, 2, retriesForFiles([native])).join(' ')).toContain('--retry 1');
|
||||
expect(retriesForFiles([native])).toBe(0);
|
||||
expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(0);
|
||||
expect(buildPaidShardArgs([native], 1_800_000, 2, retriesForFiles([native])).join(' ')).toContain('--retry 0');
|
||||
const overlay = 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts';
|
||||
expect(retriesForFiles([overlay])).toBe(0);
|
||||
});
|
||||
test('overrides exist only for the files whose matrix rows earned them, and each names a real file', () => {
|
||||
expect(Object.keys(RETRY_OVERRIDES).sort()).toEqual([
|
||||
'test/skill-e2e-office-hours-auto-mode.test.ts',
|
||||
'test/skill-e2e-plan-mode-no-op.test.ts',
|
||||
'test/skill-e2e-workflow.test.ts',
|
||||
]);
|
||||
for (const file of Object.keys(RETRY_OVERRIDES)) {
|
||||
expect(fs.existsSync(path.join(ROOT, file)), `stale RETRY_OVERRIDES entry: ${file}`).toBe(true);
|
||||
test('the matrix-era earned retries now follow the timeout-is-a-verdict rule, and each names a real file', () => {
|
||||
// These three old matrix rows earned `retries: 2`; every one has a
|
||||
// CAPTURE_LONG case, so a timed-out attempt is now their verdict.
|
||||
for (const file of ['test/skill-e2e-office-hours-auto-mode.test.ts', 'test/skill-e2e-plan-mode-no-op.test.ts', 'test/skill-e2e-workflow.test.ts']) {
|
||||
expect(fs.existsSync(path.join(ROOT, file)), `stale retry parity entry: ${file}`).toBe(true);
|
||||
expect(retriesForFiles([file])).toBe(0);
|
||||
}
|
||||
expect(retriesForFiles(['test/skill-e2e-workflow.test.ts'])).toBe(2);
|
||||
expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(1);
|
||||
expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(0);
|
||||
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1);
|
||||
expect(buildPaidShardArgs(['x'], 1000, 4, 2)).toContain('2');
|
||||
expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 1');
|
||||
});
|
||||
|
||||
+114
-1
@@ -13,6 +13,7 @@ import { describe, test, expect } from 'bun:test';
|
||||
import * as fs from 'fs';
|
||||
import * as os from 'os';
|
||||
import * as path from 'path';
|
||||
import { E2E_TIERS, E2E_TOUCHFILES } from './helpers/touchfiles';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
import {
|
||||
@@ -31,7 +32,21 @@ import {
|
||||
summarize,
|
||||
summaryExitCode,
|
||||
tierSkipReason,
|
||||
marathonSkipReason,
|
||||
CASE_SHARDED_FILES,
|
||||
CASE_TEST_NAMES,
|
||||
caseTestNamePattern,
|
||||
expandCaseShards,
|
||||
fileCaseRegistration,
|
||||
resolvePaidShardBudget,
|
||||
retriesForFiles,
|
||||
shardCaseId,
|
||||
shardFile,
|
||||
shardSlug,
|
||||
verifySliceResults,
|
||||
selectPaidTestFiles,
|
||||
buildRunManifest,
|
||||
parseRunManifest,
|
||||
type ShardOutcome,
|
||||
} from '../scripts/test-paid-shards';
|
||||
|
||||
@@ -117,7 +132,105 @@ describe('tier lane skip (B5)', () => {
|
||||
}
|
||||
const workflow = fs.readFileSync(path.join(ROOT, '.github/workflows/evals-periodic.yml'), 'utf8');
|
||||
expect(workflow.match(/--skip-judges/g)).toHaveLength(1);
|
||||
expect(workflow).toMatch(/--tier gate --emit-plan \/tmp\/gate-census-plan\/manifest\.json --slices 7 --skip-judges/);
|
||||
expect(workflow).toMatch(/--tier gate --emit-plan \/tmp\/gate-census-plan\/manifest\.json --slice-budget 540 --jobs 2 --skip-judges/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('marathon tier lane', () => {
|
||||
const file = 'test/skill-e2e-sample.test.ts';
|
||||
const reg = { 'sample-gate': [file], 'sample-long': [file] } as Record<string, string[]>;
|
||||
const tiers = { 'sample-gate': 'gate', 'sample-long': 'marathon' };
|
||||
|
||||
test('a marathon-declared file never enters the gate or periodic lane', () => {
|
||||
const source = "const describeE2E = describeE2ETier('marathon');";
|
||||
expect(classifyPaidTestFile(source, 'marathon')).toEqual({ included: true, reason: "declares tier 'marathon'" });
|
||||
for (const tier of ['gate', 'periodic'] as const) {
|
||||
expect(classifyPaidTestFile(source, tier)).toEqual({ included: false, reason: "declares tier 'marathon' only" });
|
||||
}
|
||||
expect(marathonSkipReason(file, source, {}, {})).toBeNull();
|
||||
});
|
||||
|
||||
test('marathon selects positively: only declared files or files registering a marathon case', () => {
|
||||
expect(marathonSkipReason(file, "testIfSelected('sample-long', async () => {});", reg, tiers)).toBeNull();
|
||||
expect(marathonSkipReason(file, "testIfSelected(name, async () => {});", reg, tiers)).toBeNull();
|
||||
const gateOnly = { 'sample-gate': [file] };
|
||||
for (const source of ["testIfSelected('sample-gate', async () => {});", "testIfSelected(name, async () => {});", ''])
|
||||
expect(marathonSkipReason(file, source, gateOnly, tiers)).toBe('skipped: declares no marathon tier and registers no marathon case');
|
||||
const periodic = "const describeE2E = describeE2ETier('periodic');";
|
||||
expect(classifyPaidTestFile(periodic, 'marathon')).toEqual({ included: false, reason: "declares tier 'periodic' only" });
|
||||
});
|
||||
|
||||
test('a registered marathon case keeps its gate sibling scheduled in the gate lane', () => {
|
||||
const source = "testIfSelected('sample-gate', async () => {}); testIfSelected('sample-long', async () => {});";
|
||||
expect(tierSkipReason(file, source, 'gate', reg, tiers)).toBeNull();
|
||||
expect(tierSkipReason(file, source, 'periodic', reg, tiers)).toBe('skipped: no E2E_TIERS id has tier periodic');
|
||||
});
|
||||
|
||||
test('the live marathon lane only plans files that carry marathon work, never the LLM judges', () => {
|
||||
const { selected, excluded } = selectPaidTestFiles(collectPaidTestFiles(), 'marathon');
|
||||
expect(selected).not.toContain('test/skill-llm-eval.test.ts');
|
||||
for (const file of selected) {
|
||||
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
|
||||
expect(marathonSkipReason(file, source), file).toBeNull();
|
||||
}
|
||||
expect(selected.length + excluded.length).toBe(collectPaidTestFiles().length);
|
||||
});
|
||||
});
|
||||
|
||||
describe('case-sharded files', () => {
|
||||
// Paid cases only: `if (!evalsEnabled) test(...)` blocks are free checks.
|
||||
const caseNames = (source: string) => [...source.matchAll(
|
||||
/(?<![.\w])(?<!if \(!evalsEnabled\) )(?:testConcurrentIfSelected|testIfSelected|test(?:\.serial|\.concurrent)?)\(\s*(['"])(.+?)\1/g,
|
||||
)].map(match => match[2]!);
|
||||
|
||||
for (const file of CASE_SHARDED_FILES) {
|
||||
test(`${file}: every Bun case is a registered E2E case, so every case gets a shard`, () => {
|
||||
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
|
||||
const { registered, known } = fileCaseRegistration(file, source);
|
||||
expect(known).toBe(true);
|
||||
expect(caseNames(source).sort()).toEqual(registered.map(id => CASE_TEST_NAMES[id] ?? id).sort());
|
||||
const keys = (['gate', 'periodic', 'marathon'] as const).flatMap(tier => expandCaseShards([file], tier));
|
||||
expect(keys.map(key => shardCaseId(key)).sort()).toEqual([...registered].sort());
|
||||
for (const key of keys) expect(shardFile(key)).toBe(file);
|
||||
});
|
||||
}
|
||||
|
||||
test('a case key runs exactly its case: exact name pattern, own eval slug, per-case supervision', () => {
|
||||
const pattern = new RegExp(caseTestNamePattern(['design-review-detector-shim']));
|
||||
expect(pattern.test('Design review detector shim E2E design-review-detector-shim')).toBe(true);
|
||||
expect(pattern.test('Design review detector shim E2E design-review-detector-shim-dom')).toBe(false);
|
||||
expect(new RegExp(caseTestNamePattern(['plan-review-report'])).test('Plan Review Report E2E /plan-eng-review writes GSTACK REVIEW REPORT to plan file')).toBe(true);
|
||||
const key = 'test/skill-e2e-plan.test.ts#plan-ceo-review';
|
||||
expect(shardSlug([key])).toBe('skill-e2e-plan--plan-ceo-review');
|
||||
expect(shardSlug([key])).not.toBe(shardSlug(['test/skill-e2e-plan.test.ts#plan-eng-review']));
|
||||
expect(retriesForFiles([key])).toBe(retriesForFiles(['test/skill-e2e-plan.test.ts']));
|
||||
const whole = resolvePaidShardBudget(['test/skill-e2e-plan.test.ts']);
|
||||
const one = resolvePaidShardBudget([key]);
|
||||
expect(one.policyId).toBe(whole.policyId);
|
||||
expect(one.timeoutMs).toBeLessThan(whole.timeoutMs);
|
||||
expect(planPaidShards([key, 'test/skill-e2e-plan.test.ts#plan-eng-review', 'test/a.test.ts'], { maxFilesPerShard: 3 }))
|
||||
.toEqual([['test/a.test.ts'], [key], ['test/skill-e2e-plan.test.ts#plan-eng-review']]);
|
||||
});
|
||||
|
||||
test('manifests plan each case once and results must execute exactly that case', () => {
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceBudgetMs: 540_000, jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
const planned = manifest.entries.filter(entry => entry.status === 'planned');
|
||||
for (const file of CASE_SHARDED_FILES) {
|
||||
expect(planned.some(entry => entry.file === file)).toBe(false);
|
||||
const gateIds = Object.keys(E2E_TOUCHFILES).filter(id => E2E_TOUCHFILES[id]!.includes(file) && E2E_TIERS[id] === 'gate');
|
||||
expect(planned.filter(entry => shardFile(entry.file) === file).map(entry => shardCaseId(entry.file)).sort()).toEqual(gateIds.sort());
|
||||
}
|
||||
const results = Array.from({ length: manifest.sliceCount }, (_, i) => ({ version: 1 as const, tier: 'gate' as const, sliceIndex: i + 1, sliceCount: manifest.sliceCount,
|
||||
outcomes: planned.filter(entry => entry.slice === i + 1).map(entry => ({ files: [entry.file], status: 'passed' as const, exitCode: 0, elapsedMs: 1,
|
||||
executedTests: shardCaseId(entry.file) ? 3 : 1, skippedTests: shardCaseId(entry.file) ? 2 : 0, ...(entry.budget ? { budget: entry.budget } : {}) })) }));
|
||||
expect(verifySliceResults(manifest, results).problems.filter(problem => problem.includes('Case shard'))).toEqual([]);
|
||||
const empty = structuredClone(results);
|
||||
const victim = empty.flatMap(result => result.outcomes).find(outcome => shardCaseId(outcome.files[0]!))!;
|
||||
victim.skippedTests = victim.executedTests;
|
||||
expect(verifySliceResults(manifest, empty).problems).toContain(`Case shard must execute exactly its one case: ${victim.files[0]}`);
|
||||
const whole: any = structuredClone(manifest);
|
||||
whole.entries.push({ file: CASE_SHARDED_FILES[0], slice: 1, status: 'planned', estimatedMs: 1 });
|
||||
expect(() => parseRunManifest(JSON.stringify(whole))).toThrow('one registered case per shard');
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
Reference in new issue
Block a user