name: E2E Evals on: pull_request: branches: [main] workflow_dispatch: inputs: evals_all: description: 'Run ALL gate tests in the sliced lane (bypass diff selection; also arms the hollow-shard guard)' type: boolean default: true validation_phase: description: 'Validation branch phase; run quality before behavior on unchanged inputs' type: choice options: [all, quality, cookie-quality, behavior, cookie-behavior] default: all concurrency: group: evals-${{ github.event.pull_request.number || github.run_id }} cancel-in-progress: true env: IMAGE: ghcr.io/${{ github.repository }}/ci # PRs run changed fast probes; manual runs retain the complete gate census. EVALS_PROFILE: ${{ github.event_name == 'pull_request' && 'pr' || 'full' }} EVALS_FRESH: ${{ github.event_name == 'workflow_dispatch' && '1' || '' }} jobs: # Build Docker image with pre-baked toolchain (cached — only rebuilds on Dockerfile/lockfile change) build-image: # Dependabot-triggered pull_request runs get a read-only GITHUB_TOKEN, so # a lockfile bump = new hash = failed ghcr push = permanently red check # (EV6, fork port wave 2). Skip the build for dependabot; the evals job's # explicit actor guard mirrors it because no eval test selects on a lockfile-only # diff — a maintainer's next push rebuilds the image with real perms. if: github.actor != 'dependabot[bot]' runs-on: ubicloud-standard-8 timeout-minutes: 15 permissions: contents: read packages: write outputs: image-tag: ${{ steps.meta.outputs.tag }} runtime-id: ${{ steps.runtime.outputs.id }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 - id: meta # Key on Dockerfile + lockfile only. package.json is deliberately NOT # hashed: its version field changes on every ship (60/60 recent commits), # which rebuilt the image each time for a dependency set that only # bun.lock determines. A stale baked package.json is harmless — checkout # overwrites /workspace and node_modules comes from the lockfile. run: echo "tag=${{ env.IMAGE }}:${{ hashFiles('.github/docker/Dockerfile.ci', 'bun.lock', 'patches/**') }}" >> "$GITHUB_OUTPUT" - uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4 with: registry: ghcr.io username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} - name: Check if image exists id: check run: | if docker manifest inspect ${{ steps.meta.outputs.tag }} > /dev/null 2>&1; then echo "exists=true" >> "$GITHUB_OUTPUT" else echo "exists=false" >> "$GITHUB_OUTPUT" fi - if: steps.check.outputs.exists == 'false' run: cp package.json bun.lock .github/docker/ && cp -R patches .github/docker/patches # A fork PR's GITHUB_TOKEN only has `packages: read`, so pushing fails. # Still BUILD (validates Dockerfile.ci changes), just don't publish. This # job intentionally keeps no `if:` so fork PRs still get one real, honest # green check here instead of a run where every job is grey. # Registry cache export needs a docker-container builder — the default # `docker` driver hard-errors on cache-to (first live run of the trio). - if: steps.check.outputs.exists == 'false' uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4 - if: steps.check.outputs.exists == 'false' uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7 with: context: .github/docker file: .github/docker/Dockerfile.ci push: ${{ github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository }} # Registry layer cache: reads are safe everywhere; the export is gated # to same-repo runs because a fork PR's token can't write GHCR. cache-from: type=registry,ref=${{ env.IMAGE }}:buildcache cache-to: ${{ (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && format('type=registry,ref={0}:buildcache,mode=max', env.IMAGE) || '' }} tags: | ${{ steps.meta.outputs.tag }} ${{ env.IMAGE }}:latest - name: Identify the installed eval runtime id: runtime if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository env: EVAL_IMAGE: ${{ steps.meta.outputs.tag }} run: | docker manifest inspect "$EVAL_IMAGE" > /tmp/eval-runtime-manifest.json echo "id=$(sha256sum /tmp/eval-runtime-manifest.json | cut -d ' ' -f1)" >> "$GITHUB_OUTPUT" # ── Sliced lane (the ONLY paid lane; legacy 17-row matrix deleted) ────────── # One PLANNER computes diff selection + the slice plan ONCE (killing # per-slice selector divergence); K executors consume the manifest; the # report reconciles results against it FAIL-CLOSED (a slice whose artifact # never landed is a failure, a planned shard nobody reported is a failure — # hollow lanes cannot aggregate green). Engine: scripts/test-paid-shards.ts — # the same runner local eval:bg:gate uses, so CI and local share one # selection engine, and every gate-tier file is in the census by # construction (no hand-enumerated rows to drift). The legacy matrix ran # 18 enumerated files for 22.6 min/$21 per PR serialized AHEAD of this # lane's 49-file diff-selected census; parity was demonstrated (sliced # census ⊇ matrix files) and the matrix deleted — one revert restores it. # # Fork PRs never receive repository secrets (ANTHROPIC_API_KEY et al), so # every API-calling eval fails at SDK auth before a model runs. Skip # deterministically; fork work gets real coverage via a trusted base-repo # branch (see CLAUDE.md's garrytan-agents workflow). plan-slices: runs-on: ubicloud-standard-8 if: github.actor != 'dependabot[bot]' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) timeout-minutes: 10 permissions: contents: read outputs: slices: ${{ steps.matrix.outputs.slices }} timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: # Preserve full history for merge-base diff selection. Moving the # planner off the eval image must not change its selection inputs. fetch-depth: 0 persist-credentials: false - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 with: bun-version: 1.4.0 # Planner-side reuse: restore this PR's newest receipt store (the report # job saves one merged store per run) and ship ONE filtered set with the # plan, so every trial of a panel sees the same receipts and a newer FAIL # blocks any older PASS for the same inputs. - name: Restore this PR's verified judge and E2E results if: github.event_name == 'pull_request' uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 with: path: /tmp/gstack-eval-input-cache key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-plan restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}- - name: Emit run manifest if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all' env: EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }} EVALS_CACHE_DIR: ${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }} run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 16 - name: Emit validation-phase manifest if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all' env: VALIDATION_PHASE: ${{ inputs.validation_phase }} EVALS_ALL: ${{ inputs.evals_all && '1' || '' }} EVALS_TIER: gate run: | bun --no-install -e ' import { mkdirSync, writeFileSync } from "node:fs"; import { buildRunManifest, collectPaidTestFiles, restrictManifestSelection } from "./scripts/test-paid-shards.ts"; const phase = process.env.VALIDATION_PHASE; if (!["quality", "cookie-quality", "behavior", "cookie-behavior"].includes(phase)) throw new Error("Invalid validation phase"); const cookieBehavior = phase === "cookie-behavior"; const discovered = phase === "cookie-quality" ? ["test/skill-llm-eval.test.ts"] : cookieBehavior ? ["test/skill-e2e-bws.test.ts", "test/skill-e2e-qa-workflow.test.ts", "test/skill-e2e-design.test.ts", "test/skill-e2e-diagram.test.ts", "test/skill-e2e-deploy.test.ts"] : collectPaidTestFiles().filter(file => file.startsWith("test/skill-llm-eval") === (phase === "quality")); const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceBudgetMs: 540000, jobs: 2, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered, ...(cookieBehavior ? { changedFiles: ["browse/src/cookie-picker-routes.ts", "browse/src/cookie-import-browser.ts", "browse/src/bun-polyfill.cjs"], env: { ...process.env, EVALS_ALL: "" } } : {}) }); const subset = phase === "cookie-quality" ? { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] } : cookieBehavior ? { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] } : null; const restricted = subset ? restrictManifestSelection(manifest, subset, "outside the " + phase + " validation subset") : manifest; restricted.selectionReason = phase + " validation subset; " + manifest.selectionReason; mkdirSync("/tmp/paid-plan", { recursive: true }); writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(restricted, null, 2) + "\n"); console.log(phase + ": " + restricted.entries.filter(entry => entry.status === "planned").length + " planned shards"); ' - name: Derive the executor matrix from the plan id: matrix run: | echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT" echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT" - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: paid-plan path: | /tmp/paid-plan/manifest.json /tmp/paid-plan/receipts retention-days: 30 eval-slices: runs-on: ubicloud-standard-8 needs: [build-image, plan-slices] env: EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }} # !cancelled(), not always(): still runs when build-image was skipped # (image already published), but a newer push's cancel-in-progress stops # it instead of letting a superseded run finish its paid slices first. if: ${{ !cancelled() && needs.build-image.result == 'success' && needs.plan-slices.result == 'success' }} # The planner packs ~9 minutes of recorded work per slice (EVALS_JOBS=2 x # EVALS_CONCURRENCY=2 per runner, never the old 40-way per-row fan-out # that queued claude session STARTUP behind 39 siblings). The job timeout # is the plan's supervised worst case plus 20 minutes setup/upload. timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }} permissions: contents: read packages: read container: image: ${{ needs.build-image.outputs.image-tag }} credentials: username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} options: --user runner strategy: fail-fast: false max-parallel: 16 matrix: slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: # Full history: files with SELF-derived selection (the LLM-judge # map, routing) walk git at module load, and selection is # fail-closed on git errors — a shallow checkout crashed those # shards on the lane's first live run ("ambiguous argument # 'main...HEAD'"). The manifest still governs WHICH shards run. fetch-depth: 0 persist-credentials: false - name: Fix bun temp uses: ./.github/actions/fix-bun-temp - name: Restore deps uses: ./.github/actions/restore-deps - run: bun run build # Any slice can host a PTY smoke, so the seed/registration steps run # UNCONDITIONALLY (both are idempotent) — the old matrix keyed them on # matrix.suite.name, which a sliced lane cannot do. The register # composite carries the fail-fast dangling-symlink/frontmatter # verification loop, so a moved skill target fails HERE in seconds, # not as a wedged PTY session at the shard wall. - name: Seed claude interactive config uses: ./.github/actions/seed-claude-config with: anthropic-api-key: ${{ secrets.ANTHROPIC_API_KEY }} - name: Register gstack skills for PTY smokes uses: ./.github/actions/register-gstack-skills - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: name: paid-plan path: /tmp/paid-plan # Receipts come only from the plan (this PR's store, filtered once by the # planner); new receipts land beside the slice results and the report # merges them into the next store. - name: Seed this slice's receipts from the plan run: | mkdir -p /tmp/paid-slice-results/receipts if [ -d /tmp/paid-plan/receipts ]; then cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/; fi - name: Run slice ${{ matrix.slice }} env: ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }} PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers EVALS_JOBS: "2" EVALS_CONCURRENCY: "2" GSTACK_EVAL_DIR: /tmp/paid-slice-results EVALS_CACHE_DIR: /tmp/paid-slice-results/receipts EVALS_CACHE_REPOSITORY: ${{ github.repository }} EVALS_CACHE_PR: ${{ github.event.pull_request.number }} EVALS_CACHE_RUNTIME_ID: ${{ needs.build-image.outputs.runtime-id }} run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/paid-plan/manifest.json --slice ${{ matrix.slice }} # Attempt-scoped: a re-run attempt's trials are reported under that # attempt and never replace (or collide with) the first attempt's. - name: Upload slice results if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }} path: /tmp/paid-slice-results retention-days: 90 - name: Upload native capture evidence if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: native-captures-${{ env.EVALS_RUN_ID }} include-hidden-files: true path: | ~/.gstack/projects/*/e2e-runs ~/.gstack/projects/*/evals/qa-callers ~/.gstack-dev/e2e-runs ~/.gstack-dev/evals/qa-callers if-no-files-found: ignore retention-days: 90 # The spooled per-shard full logs — a red weekly/PR lane three weeks # later needs more than a summary line. # always(), not failure(): a failed behavior trial is a verdict and no # longer reds its runner, but its full log is the diagnosis evidence. - name: Upload shard logs if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }} include-hidden-files: true # The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the # runner's spool lands THERE, not /tmp — the original /tmp glob # uploaded nothing and a red slice's diagnostics were unreachable. path: | /home/runner/.cache/gstack-paid-shard-*.log /tmp/gstack-paid-shard-*.log if-no-files-found: ignore retention-days: 30 slices-report: runs-on: ubicloud-standard-2 needs: [plan-slices, eval-slices] # !cancelled(): the report must run (and FAIL) when an executor died — a # missing slice artifact reading as green is the class this lane kills — # but a run superseded by a newer push stops here. if: ${{ !cancelled() && needs.plan-slices.result == 'success' }} timeout-minutes: 5 # contents:read ONLY — this job executes the PR-authored reconcile # runner from the PR checkout, so it # must never hold a write-scoped token. The PR comment lives in the # separate slices-comment job below, which runs NO repo code: a # $GITHUB_ENV/BASH_ENV persistence trick is job-scoped, so the split is # the trust boundary (codex adversarial finding, 2026-08-31 — the old # matrix-era report job had this separation and the consolidation had # regressed it). permissions: contents: read outputs: reconcile-exit: ${{ steps.reconcile.outputs.exit }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: persist-credentials: false - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 with: bun-version: 1.4.0 - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: name: paid-plan path: /tmp/paid-report # One directory per attempt-scoped slice artifact (no merge): shard # records can never overwrite each other, and the report keeps the # first attempt's verdict. - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: pattern: paid-slice-* path: /tmp/paid-report - name: Reconcile slices against the manifest (fail-closed) id: reconcile run: | set +e EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --report /tmp/paid-report | tee /tmp/report.txt # PIPESTATUS[0], NOT $?: GitHub's default run-step shell is # `bash -e {0}` with NO pipefail, so $? after the pipe is tee's # exit (always 0) — the fail-closed gate was silently fail-open # (caught by the ship review army; the wiring test now pins this). echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT" - name: Stamp trial history series if: always() run: | if [ -f /tmp/paid-report/trial-outcomes.jsonl ]; then bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl fi # One merged receipt store per run: the plan's shipped set, every slice's # new pass receipts, and the report's panel and negative receipts. Saved # last, so the next planner restores it as the newest prefix match. - name: Merge this run's receipts if: always() && github.event_name == 'pull_request' run: | bun --no-install run scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache \ /tmp/paid-report/receipts /tmp/paid-report/report-receipts /tmp/paid-report/paid-slice-*/receipts - name: Save this PR's verified judge and E2E results if: always() && github.event_name == 'pull_request' uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 with: path: /tmp/gstack-eval-input-cache key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged - name: Upload reconciliation output for the comment job if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: report-verdict-a${{ github.run_attempt }} path: | /tmp/report.txt /tmp/paid-report/collector-outcomes.json /tmp/paid-report/report-summary.md if-no-files-found: ignore retention-days: 30 - name: Upload trial outcomes for pass-rate history if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: trial-outcomes-pr-a${{ github.run_attempt }} path: /tmp/paid-report/trial-outcomes.jsonl if-no-files-found: ignore retention-days: 90 - name: Fail the workflow when reconciliation failed if: steps.reconcile.outputs.exit != '0' run: exit 1 # PR comment in its OWN job with the write token and ZERO repo code: no # checkout, no bun install — only downloaded artifacts, jq, and gh. See the # trust-boundary note on slices-report. slices-comment: runs-on: ubicloud-standard-2 needs: slices-report if: ${{ !cancelled() && github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && needs.slices-report.result != 'skipped' }} timeout-minutes: 5 permissions: pull-requests: write # The comment upsert calls the REST `/issues/{n}/comments` endpoints # (gh api ... issues/comments). With GITHUB_TOKEN those are gated by the # `issues` permission, not `pull-requests` (#1802 CI fix). issues: write steps: - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: name: paid-plan path: /tmp/paid-report - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: name: report-verdict-a${{ github.run_attempt }} path: /tmp/verdict continue-on-error: true # Every count, verdict and failure line comes from the read-only report # job's collector-outcomes v2 (panelVerdict() ran there); this job runs # no repo code and never recomputes a verdict. Keeps the # "## E2E Evals" marker so the upsert keeps updating the same comment. # Runs even when reconciliation failed — a red lane on the PR is the point. - name: Post PR comment env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} RECONCILE_EXIT: ${{ needs.slices-report.outputs.reconcile-exit }} run: | # shellcheck disable=SC2086,SC2059 TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; EXECUTED=0; REUSED=0; COST="0" SUITE_LINES="" VERIFIED=/tmp/verdict/paid-report/collector-outcomes.json if ! jq -e ' . as $summary | .version == 2 and (.files | type == "array") and (.totals | type == "object") and (.headline | type == "array") and (.failures | type == "array") and (.panels | type == "array") and (.verdict.verdict == "GREEN" or .verdict.verdict == "RED") and ([.files[] | .total == (.passed + .failed + .manual_accepted) and (.total == (.executed + .reused)) and ([.total,.passed,.failed,.manual_accepted,.executed,.reused,.attempts,.flaky] | all(. >= 0 and (floor == .))) ] | all) and (.totals | .total == (.passed + .failed + .manual_accepted) and .total == (.executed + .reused)) and (["total","passed","failed","manual_accepted","executed","reused","attempts","flaky"] | all(. as $key | ([$summary.files[] | .[$key]] | add // 0) == $summary.totals[$key])) ' "$VERIFIED" >/dev/null 2>&1; then VERIFIED="" echo 'Verified report summary unavailable; no verdict, counts or manual acceptance can be shown.' fi HEADLINE='(no verified report headline)' FAILURES="" if [ -n "$VERIFIED" ]; then while IFS=$'\t' read -r f T P F M FL EX RE _ATTEMPTS C TIER SHARD; do [ "$T" -eq 0 ] && continue TOTAL=$((TOTAL + T)); PASSED=$((PASSED + P)); FAILED=$((FAILED + F)) MANUAL=$((MANUAL + M)) EXECUTED=$((EXECUTED + EX)); REUSED=$((REUSED + RE)) COST=$(echo "$COST + $C" | bc) STATUS_ICON="✅" [ "$M" -gt 0 ] && STATUS_ICON="⚠ manual/unscored" [ "$F" -gt 0 ] && STATUS_ICON="❌" SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | ${M} | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n" done < <(jq -r '.files[] | [.file,.total,.passed,.failed,.manual_accepted,.flaky,.executed,.reused,.attempts,.cost,.tier,.shard] | @tsv' "$VERIFIED") # Report-sanitized lines (no @-mentions, one capped line each), fenced here. HEADLINE=$(jq -r '.headline[]' "$VERIFIED") FAILURES=$(jq -r '.failures[]' "$VERIFIED") fi COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.' STATUS="✅ PASS" if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ] \ || { [ -n "$VERIFIED" ] && [ "$(jq -r '.verdict.verdict' "$VERIFIED")" != "GREEN" ]; }; then STATUS="❌ FAIL"; fi if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (verified report unavailable)'; fi BODY="## E2E Evals: ${STATUS} \`\`\` ${HEADLINE} \`\`\` **${EXECUTED} executed, ${REUSED} reused** rule/judge records | **${MANUAL} manual accepted (unscored; no score-cache credit)** | **\$${COST}** rule/judge cost | reconcile exit: ${RECONCILE_EXIT:-missing} ${COVERAGE}
Rule and judge shards | Shard | Automated result | Manual/unscored | Executed | Reused | Status | Cost | |-------|------------------|-----------------|----------|--------|--------|------| $(echo -e "$SUITE_LINES")
Fail-closed reconciliation \`\`\` $(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo '(no reconciliation output)') \`\`\`
--- *Sliced lane: planner → duration-packed executors → fail-closed report. Behavior cases run a pre-registered 3-trial panel (PASS at 2/3 with no contract violation); a PASS 2/3 is shown with its failed trial, never as a clean pass. Reused results retain their original provenance and expiry.*" if [ -n "$FAILURES" ]; then BODY="${BODY} ### Failures and split verdicts \`\`\` ${FAILURES} \`\`\`" fi COMMENT_ID=$(gh api repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/comments \ --jq '.[] | select(.body | startswith("## E2E Evals")) | .id' | tail -1) if [ -n "$COMMENT_ID" ]; then gh api "repos/${{ github.repository }}/issues/comments/${COMMENT_ID}" \ -X PATCH -f body="$BODY" else # REST, not gh's pr-comment subcommand: this job runs with NO # checkout (the token/exec split), and that subcommand resolves # the repo FROM git — it dies with "not a git repository" here. gh api "repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/comments" \ -X POST -f body="$BODY" fi