mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
feat(evals): ~12-minute blocking paid lanes and a non-blocking marathon lane
- Planner budget mode (--slice-budget S --jobs J): recorded per-tier wall times pack into as many ~9-minute executors as the work needs; the plan records per-slice estimates and the CI job timeout (supervised worst case + 20 min). evals.yml and evals-periodic.yml derive matrix size and timeout-minutes from it; max-parallel covers every slice at once. - Case shards: plan/design/review-army/shared-libs(-paths) run one registered case per process (<file>#<case id>, exact name pattern, exactly one case). - Retry rule: a timed-out attempt is a verdict. Only files whose every case budget is CAPTURE tier or shorter keep one retry; walls shrink to match. - Marathon tier: positive selection, excluded from gate/periodic planners, run by the new evals-marathon.yml (weekly + dispatch, fresh, own report). - PR-lane E2E reuse of verified first-attempt passes on identical inputs; the report rejects reuse outside the fast PR profile. - Duration seed from census run 36385945043, per tier and per case shard.
This commit is contained in:
1 parent
acb7bc02b1
commit
0023d011a3
18 files changed
+1773
-450
No files matched your search
@@ -0,0 +1,298 @@
|
||||
name: Marathon Evals
|
||||
# The NON-BLOCKING marathon lane: complete start-to-finish flows (tier
|
||||
# 'marathon' in test/helpers/touchfiles-data.ts / describeE2ETier('marathon'))
|
||||
# that take longer than a blocking lane's ~12-minute wall. They never run in
|
||||
# the PR gate (evals.yml) or the weekly periodic + gate census
|
||||
# (evals-periodic.yml); nothing requires this workflow, so a red marathon
|
||||
# reports through its own tracking issue without gating any merge. Same engine
|
||||
# and FAIL-CLOSED report as the other lanes: one planner manifest, one file per
|
||||
# runner, a missing slice artifact is a failure. Always fresh: no result reuse.
|
||||
on:
|
||||
schedule:
|
||||
- cron: '0 12 * * 6' # Saturday 12:00 UTC, clear of the Monday periodic census
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
group: evals-marathon
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
IMAGE: ghcr.io/${{ github.repository }}/ci
|
||||
EVALS_PROFILE: full
|
||||
EVALS_FRESH: "1"
|
||||
EVALS_CACHE_PURPOSE: marathon
|
||||
|
||||
jobs:
|
||||
IMAGE: ghcr.io/${{ github.repository }}/ci
|
||||
EVALS_PROFILE: full
|
||||
EVALS_FRESH: "1"
|
||||
EVALS_CACHE_PURPOSE: periodic
|
||||
|
||||
jobs:
|
||||
build-image:
|
||||
runs-on: ubicloud-standard-8
|
||||
timeout-minutes: 15
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
outputs:
|
||||
image-tag: ${{ steps.meta.outputs.tag }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
|
||||
- id: meta
|
||||
# Keep in sync with evals.yml and evals-periodic.yml — key on Dockerfile + lockfile only
|
||||
# (package.json's version field would bust the key on every ship).
|
||||
# Byte-identity pinned by test/ci-image-tag-binding.test.ts.
|
||||
run: echo "tag=${{ env.IMAGE }}:${{ hashFiles('.github/docker/Dockerfile.ci', 'bun.lock', 'patches/**') }}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Check if image exists
|
||||
id: check
|
||||
run: |
|
||||
if docker manifest inspect ${{ steps.meta.outputs.tag }} > /dev/null 2>&1; then
|
||||
echo "exists=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "exists=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- if: steps.check.outputs.exists == 'false'
|
||||
run: cp package.json bun.lock .github/docker/ && cp -R patches .github/docker/patches
|
||||
|
||||
# Registry cache export needs a docker-container builder — the default
|
||||
# `docker` driver hard-errors on cache-to.
|
||||
- if: steps.check.outputs.exists == 'false'
|
||||
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4
|
||||
|
||||
- if: steps.check.outputs.exists == 'false'
|
||||
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7
|
||||
with:
|
||||
context: .github/docker
|
||||
file: .github/docker/Dockerfile.ci
|
||||
push: true
|
||||
# Cron-triggered in the base repo only, so cache export is always safe here.
|
||||
cache-from: type=registry,ref=${{ env.IMAGE }}:buildcache
|
||||
cache-to: type=registry,ref=${{ env.IMAGE }}:buildcache,mode=max
|
||||
tags: |
|
||||
${{ steps.meta.outputs.tag }}
|
||||
${{ env.IMAGE }}:latest
|
||||
|
||||
|
||||
plan-slices:
|
||||
runs-on: ubicloud-standard-8
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
slices: ${{ steps.matrix.outputs.slices }}
|
||||
timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
|
||||
with:
|
||||
bun-version: 1.4.0
|
||||
|
||||
# One marathon file per runner: a 1-second budget never packs two
|
||||
# recorded files together.
|
||||
- name: Emit run manifest (ALL marathon tests)
|
||||
env:
|
||||
EVALS_ALL: "1"
|
||||
run: EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --emit-plan /tmp/marathon-plan/manifest.json --slice-budget 1 --jobs 1
|
||||
|
||||
- name: Derive the executor matrix from the plan
|
||||
id: matrix
|
||||
run: |
|
||||
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: marathon-plan
|
||||
path: /tmp/marathon-plan/manifest.json
|
||||
retention-days: 30
|
||||
|
||||
eval-slices:
|
||||
runs-on: ubicloud-standard-8
|
||||
needs: [build-image, plan-slices]
|
||||
env:
|
||||
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-marathon-${{ matrix.slice }}
|
||||
# One marathon file per runner; the job timeout is the plan's supervised
|
||||
# worst case plus 20 minutes setup/upload.
|
||||
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
container:
|
||||
image: ${{ needs.build-image.outputs.image-tag }}
|
||||
credentials:
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
options: --user runner
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 8
|
||||
matrix:
|
||||
slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
# Full history: files with SELF-derived selection (the LLM-judge
|
||||
# map, routing) walk git at module load, and selection is
|
||||
# fail-closed on git errors — a shallow checkout crashed those
|
||||
# shards on the lane's first live run ("ambiguous argument
|
||||
# 'main...HEAD'"). The manifest still governs WHICH shards run.
|
||||
fetch-depth: 0
|
||||
persist-credentials: false
|
||||
|
||||
- name: Fix bun temp
|
||||
uses: ./.github/actions/fix-bun-temp
|
||||
|
||||
- name: Restore deps
|
||||
uses: ./.github/actions/restore-deps
|
||||
|
||||
- run: bun run build
|
||||
|
||||
# Any slice can host a PTY test — seed + registration run
|
||||
# unconditionally (idempotent; mirrors evals.yml's sliced lane). The
|
||||
# register composite carries the fail-fast dangling-symlink/frontmatter
|
||||
# verification loop — this lane previously LACKED it, so a moved skill
|
||||
# target surfaced as a silent "Unknown command" + wedged PTY session.
|
||||
- name: Seed claude interactive config
|
||||
uses: ./.github/actions/seed-claude-config
|
||||
with:
|
||||
anthropic-api-key: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
|
||||
- name: Register gstack skills for PTY tests
|
||||
uses: ./.github/actions/register-gstack-skills
|
||||
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
||||
with:
|
||||
name: marathon-plan
|
||||
path: /tmp/marathon-plan
|
||||
|
||||
- name: Run marathon slice ${{ matrix.slice }}
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
||||
PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers
|
||||
EVALS_JOBS: "1"
|
||||
EVALS_CONCURRENCY: "2"
|
||||
GSTACK_EVAL_DIR: /tmp/marathon-slice-results
|
||||
run: EVALS_TIER=marathon bun run scripts/test-paid-shards.ts --tier marathon --plan /tmp/marathon-plan/manifest.json --slice ${{ matrix.slice }}
|
||||
|
||||
- name: Upload slice results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: marathon-slice-${{ matrix.slice }}
|
||||
path: /tmp/marathon-slice-results
|
||||
retention-days: 90
|
||||
|
||||
- name: Upload native capture evidence
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: native-captures-${{ env.EVALS_RUN_ID }}
|
||||
include-hidden-files: true
|
||||
path: |
|
||||
~/.gstack/projects/*/e2e-runs
|
||||
~/.gstack/projects/*/evals/qa-callers
|
||||
~/.gstack-dev/e2e-runs
|
||||
~/.gstack-dev/evals/qa-callers
|
||||
if-no-files-found: ignore
|
||||
retention-days: 90
|
||||
|
||||
- name: Upload shard logs on failure
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: marathon-slice-${{ matrix.slice }}-logs
|
||||
include-hidden-files: true
|
||||
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
|
||||
# runner's spool lands THERE, not /tmp — the original /tmp glob
|
||||
# uploaded nothing and a red slice's diagnostics were unreachable.
|
||||
path: |
|
||||
/home/runner/.cache/gstack-paid-shard-*.log
|
||||
/tmp/gstack-paid-shard-*.log
|
||||
if-no-files-found: ignore
|
||||
retention-days: 30
|
||||
|
||||
report:
|
||||
runs-on: ubicloud-standard-2
|
||||
needs: [plan-slices, eval-slices]
|
||||
# !cancelled(): the report must run (and FAIL) when an executor died — a
|
||||
# missing slice artifact reading as green is the class this lane kills —
|
||||
# but a cancelled run stops here.
|
||||
if: ${{ !cancelled() && needs.plan-slices.result == 'success' }}
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
issues: write
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
|
||||
with:
|
||||
bun-version: 1.4.0
|
||||
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
||||
with:
|
||||
name: marathon-plan
|
||||
path: /tmp/marathon-report
|
||||
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
||||
with:
|
||||
pattern: marathon-slice-[0-9]*
|
||||
path: /tmp/marathon-report
|
||||
merge-multiple: true
|
||||
|
||||
- name: Reconcile slices against the manifest (fail-closed)
|
||||
id: reconcile
|
||||
if: always()
|
||||
run: |
|
||||
set +e
|
||||
EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --report /tmp/marathon-report | tee /tmp/report.txt
|
||||
# PIPESTATUS[0], NOT $?: the default step shell has no pipefail.
|
||||
echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# One tracking issue for the whole lane (never one per week).
|
||||
- name: Upsert tracking issue on failure
|
||||
if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success')
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
TITLE="Weekly marathon evals: red lane needs triage"
|
||||
BODY_FILE=/tmp/issue-body.md
|
||||
{
|
||||
echo "Automated weekly marathon report (non-blocking lane) — run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
|
||||
echo
|
||||
echo "- reconciliation exit: ${{ steps.reconcile.outputs.exit }}"
|
||||
echo "- marathon slices job: ${{ needs.eval-slices.result }}"
|
||||
echo
|
||||
echo '```'
|
||||
tail -c 6000 /tmp/report.txt 2>/dev/null || echo "(no reconciliation output)"
|
||||
echo '```'
|
||||
} > "$BODY_FILE"
|
||||
EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty')
|
||||
if [ -n "$EXISTING" ]; then
|
||||
gh issue comment "$EXISTING" --repo "$GITHUB_REPOSITORY" --body-file "$BODY_FILE"
|
||||
echo "commented on #$EXISTING"
|
||||
else
|
||||
gh issue create --repo "$GITHUB_REPOSITORY" --title "$TITLE" --body-file "$BODY_FILE"
|
||||
fi
|
||||
|
||||
- name: Fail the workflow when reconciliation failed
|
||||
if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success')
|
||||
run: exit 1
|
||||
@@ -4,8 +4,13 @@ name: Periodic Evals
|
||||
# tests can't rot invisibly — the class where the autoplan-dual-voice E2E was
|
||||
# silently broken for months until a lucky local diff selected it. Engine:
|
||||
# scripts/test-paid-shards.ts (the same runner local eval:bg:periodic uses):
|
||||
# one planner manifest, 6 ordinary slices plus an overlay slice, and a FAIL-CLOSED report — a slice
|
||||
# whose artifact never landed is a failure, not an absence. The gate-census
|
||||
# one planner manifest packed by recorded durations into as many ~9-minute
|
||||
# executors as the work needs (one file, or a tightly packed group, per
|
||||
# runner; overlays share one final slice), and a FAIL-CLOSED report — a slice
|
||||
# whose artifact never landed is a failure, not an absence. The matrix size
|
||||
# and job timeout come from the plan, so they cannot drift from the census.
|
||||
# Full end-to-end flows run in the non-blocking marathon lane
|
||||
# (evals-marathon.yml), never here. The gate-census
|
||||
# job is the weekly EVALS_ALL backstop for the gate tier (PR lanes are
|
||||
# diff-billed, so without it the full gate census might never execute
|
||||
# anywhere); the hollow-shard guard (exit 0 + zero executed tests under
|
||||
@@ -84,6 +89,11 @@ jobs:
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
periodic_slices: ${{ steps.periodic-matrix.outputs.slices }}
|
||||
periodic_timeout_minutes: ${{ steps.periodic-matrix.outputs.timeout_minutes }}
|
||||
gate_slices: ${{ steps.gate-matrix.outputs.slices }}
|
||||
gate_timeout_minutes: ${{ steps.gate-matrix.outputs.timeout_minutes }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -96,7 +106,13 @@ jobs:
|
||||
- name: Emit run manifest (ALL periodic tests minus reasoned excludes)
|
||||
env:
|
||||
EVALS_ALL: "1"
|
||||
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 7
|
||||
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2
|
||||
|
||||
- name: Derive the periodic executor matrix from the plan
|
||||
id: periodic-matrix
|
||||
run: |
|
||||
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
@@ -107,7 +123,13 @@ jobs:
|
||||
- name: Emit gate census manifest (ALL gate tests)
|
||||
env:
|
||||
EVALS_ALL: "1"
|
||||
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 7 --skip-judges
|
||||
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges
|
||||
|
||||
- name: Derive the gate census executor matrix from the plan
|
||||
id: gate-matrix
|
||||
run: |
|
||||
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/gate-census-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/gate-census-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
@@ -120,9 +142,10 @@ jobs:
|
||||
needs: [build-image, plan-slices]
|
||||
env:
|
||||
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }}
|
||||
# Seven slices retain every registered case and retry. The complete
|
||||
# census needs at most 244m40s per slice, plus 20 minutes setup/upload.
|
||||
timeout-minutes: 360
|
||||
# The planner packs ~9 minutes of recorded work per slice; the job timeout
|
||||
# is its supervised worst case (every shard at its wall) plus 20 minutes
|
||||
# setup/upload, computed from the same manifest the slices execute.
|
||||
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
@@ -134,9 +157,11 @@ jobs:
|
||||
options: --user runner
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 8
|
||||
# Every planned slice starts at once; test/evals-workflow-wiring.test.ts
|
||||
# fails when the live plan outgrows this cap.
|
||||
max-parallel: 24
|
||||
matrix:
|
||||
slice: [1, 2, 3, 4, 5, 6, 7]
|
||||
slice: ${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -174,7 +199,7 @@ jobs:
|
||||
name: paid-plan
|
||||
path: /tmp/paid-plan
|
||||
|
||||
- name: Run slice ${{ matrix.slice }}/7
|
||||
- name: Run periodic slice ${{ matrix.slice }}
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
@@ -231,8 +256,8 @@ jobs:
|
||||
needs: [build-image, plan-slices]
|
||||
env:
|
||||
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-gate-census-${{ matrix.slice }}
|
||||
# Seven slices need at most 272m each, plus 20 minutes setup/upload.
|
||||
timeout-minutes: 352
|
||||
# Supervised worst case of the packed plan plus 20 minutes setup/upload.
|
||||
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.gate_timeout_minutes) }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
@@ -243,11 +268,11 @@ jobs:
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
options: --user runner
|
||||
strategy:
|
||||
# Four file workers total, each retaining two in-file case workers.
|
||||
# Two file workers per slice, each retaining two in-file case workers.
|
||||
fail-fast: false
|
||||
max-parallel: 4
|
||||
max-parallel: 16
|
||||
matrix:
|
||||
slice: [1, 2, 3, 4, 5, 6, 7]
|
||||
slice: ${{ fromJSON(needs.plan-slices.outputs.gate_slices) }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -272,13 +297,13 @@ jobs:
|
||||
name: gate-census-plan
|
||||
path: /tmp/gate-census-plan
|
||||
|
||||
- name: Run gate census slice ${{ matrix.slice }}/7
|
||||
- name: Run gate census slice ${{ matrix.slice }}
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
||||
PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers
|
||||
EVALS_JOBS: "1"
|
||||
EVALS_JOBS: "2"
|
||||
EVALS_CONCURRENCY: "2"
|
||||
GSTACK_EVAL_DIR: /tmp/gate-census-results
|
||||
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/gate-census-plan/manifest.json --slice ${{ matrix.slice }}
|
||||
|
||||
+29
-22
@@ -125,6 +125,9 @@ jobs:
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
slices: ${{ steps.matrix.outputs.slices }}
|
||||
timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -141,7 +144,7 @@ jobs:
|
||||
if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all'
|
||||
env:
|
||||
EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }}
|
||||
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slices 7
|
||||
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2
|
||||
|
||||
- name: Emit validation-phase manifest
|
||||
if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all'
|
||||
@@ -152,23 +155,30 @@ jobs:
|
||||
run: |
|
||||
bun --no-install -e '
|
||||
import { mkdirSync, writeFileSync } from "node:fs";
|
||||
import { buildRunManifest, collectPaidTestFiles } from "./scripts/test-paid-shards.ts";
|
||||
import { buildRunManifest, collectPaidTestFiles, restrictManifestSelection } from "./scripts/test-paid-shards.ts";
|
||||
const phase = process.env.VALIDATION_PHASE;
|
||||
if (!["quality", "cookie-quality", "behavior", "cookie-behavior"].includes(phase)) throw new Error("Invalid validation phase");
|
||||
const cookieBehavior = phase === "cookie-behavior";
|
||||
const discovered = phase === "cookie-quality" ? ["test/skill-llm-eval.test.ts"]
|
||||
: cookieBehavior ? ["test/skill-e2e-bws.test.ts", "test/skill-e2e-qa-workflow.test.ts", "test/skill-e2e-design.test.ts", "test/skill-e2e-diagram.test.ts", "test/skill-e2e-deploy.test.ts"]
|
||||
: collectPaidTestFiles().filter(file => file.startsWith("test/skill-llm-eval") === (phase === "quality"));
|
||||
const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceCount: 6, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered,
|
||||
const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceBudgetMs: 540000, jobs: 2, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered,
|
||||
...(cookieBehavior ? { changedFiles: ["browse/src/cookie-picker-routes.ts", "browse/src/cookie-import-browser.ts", "browse/src/bun-polyfill.cjs"], env: { ...process.env, EVALS_ALL: "" } } : {}) });
|
||||
if (phase === "cookie-quality") manifest.selection = { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] };
|
||||
if (cookieBehavior) manifest.selection = { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] };
|
||||
manifest.selectionReason = phase + " validation subset; " + manifest.selectionReason;
|
||||
const subset = phase === "cookie-quality" ? { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] }
|
||||
: cookieBehavior ? { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] } : null;
|
||||
const restricted = subset ? restrictManifestSelection(manifest, subset, "outside the " + phase + " validation subset") : manifest;
|
||||
restricted.selectionReason = phase + " validation subset; " + manifest.selectionReason;
|
||||
mkdirSync("/tmp/paid-plan", { recursive: true });
|
||||
writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(manifest, null, 2) + "\n");
|
||||
console.log(phase + ": " + manifest.entries.filter(entry => entry.status === "planned").length + " planned shards");
|
||||
writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(restricted, null, 2) + "\n");
|
||||
console.log(phase + ": " + restricted.entries.filter(entry => entry.status === "planned").length + " planned shards");
|
||||
'
|
||||
|
||||
- name: Derive the executor matrix from the plan
|
||||
id: matrix
|
||||
run: |
|
||||
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: paid-plan
|
||||
@@ -184,14 +194,11 @@ jobs:
|
||||
# (image already published), but a newer push's cancel-in-progress stops
|
||||
# it instead of letting a superseded run finish its paid slices first.
|
||||
if: ${{ !cancelled() && needs.build-image.result == 'success' && needs.plan-slices.result == 'success' }}
|
||||
# Aggregate spawn-concurrency budget: 6 slices x EVALS_JOBS=2 x
|
||||
# EVALS_CONCURRENCY=2 = 24 concurrent tests lane-wide (the old matrix's
|
||||
# 40-way per row queued claude session STARTUP behind 39 siblings and ate
|
||||
# per-test budgets — the documented timeout-flake family). Tune with
|
||||
# parity data before raising.
|
||||
# The complete gate census needs at most 242 minutes per slice; keep
|
||||
# 20 minutes for setup/upload without preempting configured retries.
|
||||
timeout-minutes: 265
|
||||
# The planner packs ~9 minutes of recorded work per slice (EVALS_JOBS=2 x
|
||||
# EVALS_CONCURRENCY=2 per runner, never the old 40-way per-row fan-out
|
||||
# that queued claude session STARTUP behind 39 siblings). The job timeout
|
||||
# is the plan's supervised worst case plus 20 minutes setup/upload.
|
||||
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: read
|
||||
@@ -203,9 +210,9 @@ jobs:
|
||||
options: --user runner
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 6
|
||||
max-parallel: 16
|
||||
matrix:
|
||||
slice: [1, 2, 3, 4, 5, 6, 7]
|
||||
slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
@@ -246,7 +253,7 @@ jobs:
|
||||
|
||||
# Only this PR's receipts are eligible. No base-branch or cross-PR restore
|
||||
# prefix; every receipt also verifies exact inputs and its original age.
|
||||
- name: Restore this PR's verified judge results
|
||||
- name: Restore this PR's verified judge and E2E results
|
||||
if: github.event_name == 'pull_request'
|
||||
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
|
||||
with:
|
||||
@@ -254,7 +261,7 @@ jobs:
|
||||
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}
|
||||
restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-
|
||||
|
||||
- name: Run slice ${{ matrix.slice }}/7
|
||||
- name: Run slice ${{ matrix.slice }}
|
||||
env:
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
@@ -285,7 +292,7 @@ jobs:
|
||||
|
||||
# An unrelated failing case does not discard already verified passes.
|
||||
# Failed/retried/partial attempts never become receipts in the first place.
|
||||
- name: Save verified judge results for this PR
|
||||
- name: Save verified judge and E2E results for this PR
|
||||
if: ${{ !cancelled() && steps.receipts.outputs.present == 'true' }}
|
||||
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
|
||||
with:
|
||||
@@ -531,7 +538,7 @@ jobs:
|
||||
</details>
|
||||
|
||||
---
|
||||
*Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → 6 executors → fail-closed report). Reused scores retain their original provenance and expiry.*"
|
||||
*Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → duration-packed executors → fail-closed report). Reused results retain their original provenance and expiry.*"
|
||||
|
||||
if [ "$FAILED" -gt 0 ]; then
|
||||
FAILURES=""
|
||||
|
||||
Reference in new issue
Block a user