Files
gstack/.github/workflows/evals-marathon.yml
T
Workflow config file is invalid. Please check your config file: yaml: construct errors: line 31: mapping key "jobs" already defined at line 25
garrytan 0023d011a3 feat(evals): ~12-minute blocking paid lanes and a non-blocking marathon lane
- Planner budget mode (--slice-budget S --jobs J): recorded per-tier wall
  times pack into as many ~9-minute executors as the work needs; the plan
  records per-slice estimates and the CI job timeout (supervised worst case
  + 20 min). evals.yml and evals-periodic.yml derive matrix size and
  timeout-minutes from it; max-parallel covers every slice at once.
- Case shards: plan/design/review-army/shared-libs(-paths) run one registered
  case per process (<file>#<case id>, exact name pattern, exactly one case).
- Retry rule: a timed-out attempt is a verdict. Only files whose every case
  budget is CAPTURE tier or shorter keep one retry; walls shrink to match.
- Marathon tier: positive selection, excluded from gate/periodic planners,
  run by the new evals-marathon.yml (weekly + dispatch, fresh, own report).
- PR-lane E2E reuse of verified first-attempt passes on identical inputs;
  the report rejects reuse outside the fast PR profile.
- Duration seed from census run 36385945043, per tier and per case shard.
2026-09-29 16:19:52 +00:00

299 lines
12 KiB
YAML

name: Marathon Evals
# The NON-BLOCKING marathon lane: complete start-to-finish flows (tier
# 'marathon' in test/helpers/touchfiles-data.ts / describeE2ETier('marathon'))
# that take longer than a blocking lane's ~12-minute wall. They never run in
# the PR gate (evals.yml) or the weekly periodic + gate census
# (evals-periodic.yml); nothing requires this workflow, so a red marathon
# reports through its own tracking issue without gating any merge. Same engine
# and FAIL-CLOSED report as the other lanes: one planner manifest, one file per
# runner, a missing slice artifact is a failure. Always fresh: no result reuse.
on:
schedule:
- cron: '0 12 * * 6' # Saturday 12:00 UTC, clear of the Monday periodic census
workflow_dispatch:
concurrency:
group: evals-marathon
cancel-in-progress: true
env:
IMAGE: ghcr.io/${{ github.repository }}/ci
EVALS_PROFILE: full
EVALS_FRESH: "1"
EVALS_CACHE_PURPOSE: marathon
jobs:
IMAGE: ghcr.io/${{ github.repository }}/ci
EVALS_PROFILE: full
EVALS_FRESH: "1"
EVALS_CACHE_PURPOSE: periodic
jobs:
build-image:
runs-on: ubicloud-standard-8
timeout-minutes: 15
permissions:
contents: read
packages: write
outputs:
image-tag: ${{ steps.meta.outputs.tag }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- id: meta
# Keep in sync with evals.yml and evals-periodic.yml — key on Dockerfile + lockfile only
# (package.json's version field would bust the key on every ship).
# Byte-identity pinned by test/ci-image-tag-binding.test.ts.
run: echo "tag=${{ env.IMAGE }}:${{ hashFiles('.github/docker/Dockerfile.ci', 'bun.lock', 'patches/**') }}" >> "$GITHUB_OUTPUT"
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Check if image exists
id: check
run: |
if docker manifest inspect ${{ steps.meta.outputs.tag }} > /dev/null 2>&1; then
echo "exists=true" >> "$GITHUB_OUTPUT"
else
echo "exists=false" >> "$GITHUB_OUTPUT"
fi
- if: steps.check.outputs.exists == 'false'
run: cp package.json bun.lock .github/docker/ && cp -R patches .github/docker/patches
# Registry cache export needs a docker-container builder — the default
# `docker` driver hard-errors on cache-to.
- if: steps.check.outputs.exists == 'false'
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4
- if: steps.check.outputs.exists == 'false'
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7
with:
context: .github/docker
file: .github/docker/Dockerfile.ci
push: true
# Cron-triggered in the base repo only, so cache export is always safe here.
cache-from: type=registry,ref=${{ env.IMAGE }}:buildcache
cache-to: type=registry,ref=${{ env.IMAGE }}:buildcache,mode=max
tags: |
${{ steps.meta.outputs.tag }}
${{ env.IMAGE }}:latest
plan-slices:
runs-on: ubicloud-standard-8
timeout-minutes: 10
permissions:
contents: read
outputs:
slices: ${{ steps.matrix.outputs.slices }}
timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
persist-credentials: false
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
with:
bun-version: 1.4.0
# One marathon file per runner: a 1-second budget never packs two
# recorded files together.
- name: Emit run manifest (ALL marathon tests)
env:
EVALS_ALL: "1"
run: EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --emit-plan /tmp/marathon-plan/manifest.json --slice-budget 1 --jobs 1
- name: Derive the executor matrix from the plan
id: matrix
run: |
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT"
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: marathon-plan
path: /tmp/marathon-plan/manifest.json
retention-days: 30
eval-slices:
runs-on: ubicloud-standard-8
needs: [build-image, plan-slices]
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-marathon-${{ matrix.slice }}
# One marathon file per runner; the job timeout is the plan's supervised
# worst case plus 20 minutes setup/upload.
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }}
permissions:
contents: read
packages: read
container:
image: ${{ needs.build-image.outputs.image-tag }}
credentials:
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
options: --user runner
strategy:
fail-fast: false
max-parallel: 8
matrix:
slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
# Full history: files with SELF-derived selection (the LLM-judge
# map, routing) walk git at module load, and selection is
# fail-closed on git errors — a shallow checkout crashed those
# shards on the lane's first live run ("ambiguous argument
# 'main...HEAD'"). The manifest still governs WHICH shards run.
fetch-depth: 0
persist-credentials: false
- name: Fix bun temp
uses: ./.github/actions/fix-bun-temp
- name: Restore deps
uses: ./.github/actions/restore-deps
- run: bun run build
# Any slice can host a PTY test — seed + registration run
# unconditionally (idempotent; mirrors evals.yml's sliced lane). The
# register composite carries the fail-fast dangling-symlink/frontmatter
# verification loop — this lane previously LACKED it, so a moved skill
# target surfaced as a silent "Unknown command" + wedged PTY session.
- name: Seed claude interactive config
uses: ./.github/actions/seed-claude-config
with:
anthropic-api-key: ${{ secrets.ANTHROPIC_API_KEY }}
- name: Register gstack skills for PTY tests
uses: ./.github/actions/register-gstack-skills
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
name: marathon-plan
path: /tmp/marathon-plan
- name: Run marathon slice ${{ matrix.slice }}
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers
EVALS_JOBS: "1"
EVALS_CONCURRENCY: "2"
GSTACK_EVAL_DIR: /tmp/marathon-slice-results
run: EVALS_TIER=marathon bun run scripts/test-paid-shards.ts --tier marathon --plan /tmp/marathon-plan/manifest.json --slice ${{ matrix.slice }}
- name: Upload slice results
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: marathon-slice-${{ matrix.slice }}
path: /tmp/marathon-slice-results
retention-days: 90
- name: Upload native capture evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: native-captures-${{ env.EVALS_RUN_ID }}
include-hidden-files: true
path: |
~/.gstack/projects/*/e2e-runs
~/.gstack/projects/*/evals/qa-callers
~/.gstack-dev/e2e-runs
~/.gstack-dev/evals/qa-callers
if-no-files-found: ignore
retention-days: 90
- name: Upload shard logs on failure
if: failure()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: marathon-slice-${{ matrix.slice }}-logs
include-hidden-files: true
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
# runner's spool lands THERE, not /tmp — the original /tmp glob
# uploaded nothing and a red slice's diagnostics were unreachable.
path: |
/home/runner/.cache/gstack-paid-shard-*.log
/tmp/gstack-paid-shard-*.log
if-no-files-found: ignore
retention-days: 30
report:
runs-on: ubicloud-standard-2
needs: [plan-slices, eval-slices]
# !cancelled(): the report must run (and FAIL) when an executor died — a
# missing slice artifact reading as green is the class this lane kills —
# but a cancelled run stops here.
if: ${{ !cancelled() && needs.plan-slices.result == 'success' }}
timeout-minutes: 10
permissions:
contents: read
issues: write
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
persist-credentials: false
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
with:
bun-version: 1.4.0
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
name: marathon-plan
path: /tmp/marathon-report
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
pattern: marathon-slice-[0-9]*
path: /tmp/marathon-report
merge-multiple: true
- name: Reconcile slices against the manifest (fail-closed)
id: reconcile
if: always()
run: |
set +e
EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --report /tmp/marathon-report | tee /tmp/report.txt
# PIPESTATUS[0], NOT $?: the default step shell has no pipefail.
echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
# One tracking issue for the whole lane (never one per week).
- name: Upsert tracking issue on failure
if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success')
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set -euo pipefail
TITLE="Weekly marathon evals: red lane needs triage"
BODY_FILE=/tmp/issue-body.md
{
echo "Automated weekly marathon report (non-blocking lane) — run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
echo
echo "- reconciliation exit: ${{ steps.reconcile.outputs.exit }}"
echo "- marathon slices job: ${{ needs.eval-slices.result }}"
echo
echo '```'
tail -c 6000 /tmp/report.txt 2>/dev/null || echo "(no reconciliation output)"
echo '```'
} > "$BODY_FILE"
EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty')
if [ -n "$EXISTING" ]; then
gh issue comment "$EXISTING" --repo "$GITHUB_REPOSITORY" --body-file "$BODY_FILE"
echo "commented on #$EXISTING"
else
gh issue create --repo "$GITHUB_REPOSITORY" --title "$TITLE" --body-file "$BODY_FILE"
fi
- name: Fail the workflow when reconciliation failed
if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success')
run: exit 1