From 8622535b90c462d5cf9c3267c12284eb0b0eb300 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:34:33 +0000 Subject: [PATCH] ci(evals): attempt-scoped artifacts, verdict-v2 PR comment, weekly pass-rate gate and one INFRA re-dispatch - Slice, census and marathon artifacts carry -a; reports download them per artifact (no merge), so records never overwrite and a re-run never replaces the first attempt's verdict. - Planners pass --max-parallel for the capacity preflight (24/16 unchanged: the refreshed periodic plan needs 24 slices, the gate census 12). - PR comment: jq-only job reads collector-outcomes v2 (headline, sanitized failure block); the group_by(.name)|last recomputation is gone. - Reports stamp series identities, upload trial-outcomes-* for history, and shard logs upload always (a failed trial no longer reds its runner). - Weekly report: headline + failure block of both lanes in the issue body, the eval:pass-rates --gate step (fails closed without history), close the issue on a green run, and UC-E1: when every red is machine-classified INFRA/INCOMPLETE, one re-dispatch as a new run in its own concurrency group (redispatch_of), both runs reported. --- .github/workflows/evals-marathon.yml | 7 +- .github/workflows/evals-periodic.yml | 180 ++++++++++++++++++++++----- .github/workflows/evals.yml | 136 +++++++++----------- test/ci-paid-coordination.test.ts | 7 +- test/evals-workflow-wiring.test.ts | 65 ++++++++++ 5 files changed, 284 insertions(+), 111 deletions(-) diff --git a/.github/workflows/evals-marathon.yml b/.github/workflows/evals-marathon.yml index b73a7ea1c..7d3e52326 100644 --- a/.github/workflows/evals-marathon.yml +++ b/.github/workflows/evals-marathon.yml @@ -187,7 +187,7 @@ jobs: if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: marathon-slice-${{ matrix.slice }} + name: marathon-slice-${{ matrix.slice }}-a${{ github.run_attempt }} path: /tmp/marathon-slice-results retention-days: 90 @@ -209,7 +209,7 @@ jobs: if: failure() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: marathon-slice-${{ matrix.slice }}-logs + name: marathon-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }} include-hidden-files: true # The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the # runner's spool lands THERE, not /tmp — the original /tmp glob @@ -247,9 +247,8 @@ jobs: - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: - pattern: marathon-slice-[0-9]* + pattern: marathon-slice-* path: /tmp/marathon-report - merge-multiple: true - name: Reconcile slices against the manifest (fail-closed) id: reconcile diff --git a/.github/workflows/evals-periodic.yml b/.github/workflows/evals-periodic.yml index a684b7b5b..707c2221c 100644 --- a/.github/workflows/evals-periodic.yml +++ b/.github/workflows/evals-periodic.yml @@ -19,9 +19,16 @@ on: schedule: - cron: '0 6 * * 1' # Monday 6 AM UTC (ci-image prebuilds at 4 AM) workflow_dispatch: + inputs: + redispatch_of: + description: 'Run id this run re-dispatches (the one INFRA/INCOMPLETE-only re-dispatch; set by the report job)' + type: string + default: '' +# A re-dispatch runs in its own group so it never cancels the run that +# dispatched it; both runs are reported. concurrency: - group: evals-periodic + group: evals-periodic${{ inputs.redispatch_of && format('-redispatch-{0}', inputs.redispatch_of) || '' }} cancel-in-progress: true env: @@ -106,7 +113,7 @@ jobs: - name: Emit run manifest (ALL periodic tests minus reasoned excludes) env: EVALS_ALL: "1" - run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 + run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 24 - name: Derive the periodic executor matrix from the plan id: periodic-matrix @@ -123,7 +130,7 @@ jobs: - name: Emit gate census manifest (ALL gate tests) env: EVALS_ALL: "1" - run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges + run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges --max-parallel 16 - name: Derive the gate census executor matrix from the plan id: gate-matrix @@ -214,7 +221,7 @@ jobs: if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: paid-slice-${{ matrix.slice }} + name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }} path: /tmp/paid-slice-results retention-days: 90 @@ -232,11 +239,13 @@ jobs: if-no-files-found: ignore retention-days: 90 - - name: Upload shard logs on failure - if: failure() + # always(), not failure(): a failed behavior trial is a verdict and no + # longer reds its runner, but its full log is the diagnosis evidence. + - name: Upload shard logs + if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: paid-slice-${{ matrix.slice }}-logs + name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }} include-hidden-files: true # The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the # runner's spool lands THERE, not /tmp — the original /tmp glob @@ -312,7 +321,7 @@ jobs: if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: gate-census-${{ matrix.slice }} + name: gate-census-${{ matrix.slice }}-a${{ github.run_attempt }} path: /tmp/gate-census-results retention-days: 90 @@ -337,12 +346,16 @@ jobs: # missing slice artifact reading as green is the class this lane kills — # but a cancelled run stops here. if: ${{ !cancelled() && needs.plan-slices.result == 'success' }} - timeout-minutes: 10 + timeout-minutes: 15 permissions: contents: read - # The failure notification below upserts a tracking issue via - # `gh api /issues` — gated by the issues permission. + # The notification below upserts (or closes) a tracking issue via + # `gh issue` — gated by the issues permission. issues: write + # Pass-rate history downloads earlier weekly runs' trial-outcomes. + actions: read + outputs: + redispatch: ${{ steps.verdict.outputs.redispatch }} steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: @@ -357,11 +370,12 @@ jobs: name: paid-plan path: /tmp/paid-report + # One directory per attempt-scoped slice artifact (no merge): shard + # records never overwrite each other and the first attempt decides. - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: - pattern: paid-slice-[0-9]* + pattern: paid-slice-* path: /tmp/paid-report - merge-multiple: true - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: @@ -372,7 +386,6 @@ jobs: with: pattern: gate-census-[0-9]* path: /tmp/gate-census-report - merge-multiple: true - name: Reconcile slices against the manifest (fail-closed) id: reconcile @@ -394,34 +407,114 @@ jobs: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --report /tmp/gate-census-report | tee /tmp/gate-report.txt echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT" - # A red weekly lane nobody must action is waste — upsert ONE tracking - # issue (never a new issue per week) with the reconciliation output, so - # failures have an owner-visible artifact with history in one place. - - name: Upsert tracking issue on failure - if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success') + - name: Stamp trial history series + if: always() + run: | + for file in /tmp/paid-report/trial-outcomes.jsonl /tmp/gate-census-report/trial-outcomes.jsonl; do + if [ -f "$file" ]; then bun --no-install run scripts/eval-trial-series.ts "$file"; fi + done + + - name: Upload trial outcomes for pass-rate history + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: trial-outcomes-periodic-a${{ github.run_attempt }} + path: | + /tmp/paid-report/trial-outcomes.jsonl + if-no-files-found: ignore + retention-days: 90 + + - name: Upload gate census trial outcomes for pass-rate history + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: trial-outcomes-gate-census-a${{ github.run_attempt }} + path: | + /tmp/gate-census-report/trial-outcomes.jsonl + if-no-files-found: ignore + retention-days: 90 + + # Weekly pass-rate gate over the last 10 weekly runs (drift, rule cases + # behaving like behavior, quarantine exit/expiry/cap). Fails closed when + # history cannot be fetched. + - name: Pass-rate history gate + id: pass-rates + if: always() env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + set +e + bun run eval:pass-rates --gate --runs 10 > /tmp/pass-rates.txt 2>&1 + echo "exit=$?" >> "$GITHUB_OUTPUT" + cat /tmp/pass-rates.txt + + # UC-E1 (approved): a run whose every red verdict is machine-classified + # INFRA or INCOMPLETE may be re-dispatched ONCE as a new run. + - name: Classify the census verdict + id: verdict + if: always() + env: + REDISPATCH_OF: ${{ inputs.redispatch_of }} + PERIODIC_EXIT: ${{ steps.reconcile.outputs.exit }} + GATE_EXIT: ${{ steps.gate-reconcile.outputs.exit }} + run: | + eligible() { # $1 exit, $2 report dir + [ "$1" = "0" ] && return 0 + jq -e '.version == 2 and .verdict.redispatchEligible == true' "$2/collector-outcomes.json" >/dev/null 2>&1 + } + if [ -z "$REDISPATCH_OF" ] && { [ "$PERIODIC_EXIT" != "0" ] || [ "$GATE_EXIT" != "0" ]; } \ + && eligible "$PERIODIC_EXIT" /tmp/paid-report && eligible "$GATE_EXIT" /tmp/gate-census-report; then + echo "redispatch=true" >> "$GITHUB_OUTPUT" + else + echo "redispatch=false" >> "$GITHUB_OUTPUT" + fi + + # A red weekly lane nobody must action is waste — upsert ONE tracking + # issue (never a new issue per week) with the headline and failure block + # of both lanes, and close it on the next green run. + - name: Upsert tracking issue on failure + if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || steps.pass-rates.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success') + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + REDISPATCH: ${{ steps.verdict.outputs.redispatch }} + REDISPATCH_OF: ${{ inputs.redispatch_of }} run: | set -euo pipefail TITLE="Weekly periodic evals: red lane needs triage" BODY_FILE=/tmp/issue-body.md + RUN_URL="${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}" { - echo "Automated weekly report — run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}" + echo "Automated weekly report — run: ${RUN_URL}" + if [ -n "$REDISPATCH_OF" ]; then echo; echo "This run is the one INFRA/INCOMPLETE re-dispatch of run ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${REDISPATCH_OF}; both runs are reported."; fi + if [ "$REDISPATCH" = "true" ]; then echo; echo "Every red verdict is machine-classified INFRA/INCOMPLETE: re-dispatching once as a new run (EVAL_POLICY.infraRedispatch). This run stays red and reported."; fi echo - echo "- periodic reconciliation exit: ${{ steps.reconcile.outputs.exit }}" - echo "- periodic slices job: ${{ needs.eval-slices.result }}" - echo "- gate census reconciliation exit: ${{ steps.gate-reconcile.outputs.exit }}" - echo "- gate census job: ${{ needs.gate-census.result }}" + echo "- periodic reconciliation exit: ${{ steps.reconcile.outputs.exit }} (slices job: ${{ needs.eval-slices.result }})" + echo "- gate census reconciliation exit: ${{ steps.gate-reconcile.outputs.exit }} (census job: ${{ needs.gate-census.result }})" + echo "- pass-rate history gate exit: ${{ steps.pass-rates.outputs.exit }}" + echo + echo "### Periodic lane" + cat /tmp/paid-report/report-summary.md 2>/dev/null || echo "(no periodic report summary)" + echo + echo "### Gate census" + cat /tmp/gate-census-report/report-summary.md 2>/dev/null || echo "(no gate census report summary)" + echo + echo "### Pass-rate history (ACTION REQUIRED)" + echo '```' + { grep -E 'ACTION REQUIRED|history unavailable' /tmp/pass-rates.txt || echo "(no pass-rate alarms)"; } | sed 's/@/@\xe2\x80\x8b/g' | head -c 6000 + echo '```' + echo + echo "
Full reconciliation output" echo echo '```' - tail -c 6000 /tmp/report.txt 2>/dev/null || echo "(no reconciliation output)" + tail -c 6000 /tmp/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo "(no reconciliation output)" echo '```' echo echo '```' - tail -c 6000 /tmp/gate-report.txt 2>/dev/null || echo "(no gate census reconciliation output)" + tail -c 6000 /tmp/gate-report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo "(no gate census reconciliation output)" echo '```' + echo "
" echo - echo "Exclusion policy: test/helpers/periodic-exclude-data.ts (every entry needs reason + tracking; removal re-activates the file next week)." + echo "Policy: EVAL_POLICY and CASE_QUARANTINE in test/helpers/periodic-exclude-data.ts; history: \`bun run eval:pass-rates\`." } > "$BODY_FILE" EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty') if [ -n "$EXISTING" ]; then @@ -431,6 +524,37 @@ jobs: gh issue create --repo "$GITHUB_REPOSITORY" --title "$TITLE" --body-file "$BODY_FILE" fi + - name: Close the tracking issue on a green run + if: always() && steps.reconcile.outputs.exit == '0' && steps.gate-reconcile.outputs.exit == '0' && steps.pass-rates.outputs.exit == '0' && needs.eval-slices.result == 'success' && needs.gate-census.result == 'success' + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + REDISPATCH_OF: ${{ inputs.redispatch_of }} + run: | + set -euo pipefail + TITLE="Weekly periodic evals: red lane needs triage" + EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty') + if [ -n "$EXISTING" ]; then + NOTE="Green weekly run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}" + if [ -n "$REDISPATCH_OF" ]; then NOTE="${NOTE} (the INFRA re-dispatch of run ${REDISPATCH_OF}, which stays red and reported)"; fi + gh issue close "$EXISTING" --repo "$GITHUB_REPOSITORY" --comment "$NOTE" + fi + - name: Fail the workflow when reconciliation failed - if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success') + if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || steps.pass-rates.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success') run: exit 1 + + # The one INFRA/INCOMPLETE re-dispatch (UC-E1). Its own job so the report + # job keeps no actions:write; the new run's concurrency group differs, so it + # never cancels this run. + redispatch: + runs-on: ubicloud-standard-2 + needs: report + if: ${{ !cancelled() && needs.report.outputs.redispatch == 'true' }} + timeout-minutes: 5 + permissions: + actions: write + steps: + - name: Re-dispatch the weekly census once + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: gh workflow run evals-periodic.yml --repo "$GITHUB_REPOSITORY" --ref "$GITHUB_REF_NAME" -f redispatch_of="$GITHUB_RUN_ID" diff --git a/.github/workflows/evals.yml b/.github/workflows/evals.yml index 87775320b..7cf43e744 100644 --- a/.github/workflows/evals.yml +++ b/.github/workflows/evals.yml @@ -144,7 +144,7 @@ jobs: if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all' env: EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }} - run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 + run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 16 - name: Emit validation-phase manifest if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all' @@ -299,11 +299,13 @@ jobs: path: /tmp/gstack-eval-input-cache key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }} + # Attempt-scoped: a re-run attempt's trials are reported under that + # attempt and never replace (or collide with) the first attempt's. - name: Upload slice results if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: paid-slice-${{ matrix.slice }} + name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }} path: /tmp/paid-slice-results retention-days: 90 @@ -323,11 +325,13 @@ jobs: # The spooled per-shard full logs — a red weekly/PR lane three weeks # later needs more than a summary line. - - name: Upload shard logs on failure - if: failure() + # always(), not failure(): a failed behavior trial is a verdict and no + # longer reds its runner, but its full log is the diagnosis evidence. + - name: Upload shard logs + if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: paid-slice-${{ matrix.slice }}-logs + name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }} include-hidden-files: true # The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the # runner's spool lands THERE, not /tmp — the original /tmp glob @@ -372,11 +376,13 @@ jobs: name: paid-plan path: /tmp/paid-report + # One directory per attempt-scoped slice artifact (no merge): shard + # records can never overwrite each other, and the report keeps the + # first attempt's verdict. - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: - pattern: paid-slice-[0-9]* + pattern: paid-slice-* path: /tmp/paid-report - merge-multiple: true - name: Reconcile slices against the manifest (fail-closed) id: reconcile @@ -389,17 +395,31 @@ jobs: # (caught by the ship review army; the wiring test now pins this). echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT" + - name: Stamp trial history series + if: always() && hashFiles('/tmp/paid-report/trial-outcomes.jsonl') != '' + run: bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl + - name: Upload reconciliation output for the comment job if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: - name: report-verdict + name: report-verdict-a${{ github.run_attempt }} path: | /tmp/report.txt /tmp/paid-report/collector-outcomes.json + /tmp/paid-report/report-summary.md if-no-files-found: ignore retention-days: 30 + - name: Upload trial outcomes for pass-rate history + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: trial-outcomes-pr-a${{ github.run_attempt }} + path: /tmp/paid-report/trial-outcomes.jsonl + if-no-files-found: ignore + retention-days: 90 + - name: Fail the workflow when reconciliation failed if: steps.reconcile.outputs.exit != '0' run: exit 1 @@ -426,18 +446,13 @@ jobs: - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 with: - pattern: paid-slice-[0-9]* - path: /tmp/paid-report - merge-multiple: true - - - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 - with: - name: report-verdict + name: report-verdict-a${{ github.run_attempt }} path: /tmp/verdict continue-on-error: true - # Verified counts come from the read-only report job, not repo code in - # this write-token job. Keeps the + # Every count, verdict and failure line comes from the read-only report + # job's collector-outcomes v2 (panelVerdict() ran there); this job runs + # no repo code and never recomputes a verdict. Keeps the # "## E2E Evals" marker so the upsert keeps updating the same comment. # Runs even when reconciliation failed — a red lane on the PR is the point. - name: Post PR comment @@ -446,13 +461,14 @@ jobs: RECONCILE_EXIT: ${{ needs.slices-report.outputs.reconcile-exit }} run: | # shellcheck disable=SC2086,SC2059 - RESULTS=$(find /tmp/paid-report -name '*.json' ! -name 'manifest.json' ! -name 'slice-*.json' ! -name '_partial*' 2>/dev/null | sort) - TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; FLAKY=0; EXECUTED=0; REUSED=0; COST="0" + TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; EXECUTED=0; REUSED=0; COST="0" SUITE_LINES="" VERIFIED=/tmp/verdict/paid-report/collector-outcomes.json if ! jq -e ' . as $summary | - .version == 1 and (.files | type == "array") and (.totals | type == "object") and + .version == 2 and (.files | type == "array") and (.totals | type == "object") and + (.headline | type == "array") and (.failures | type == "array") and (.panels | type == "array") and + (.verdict.verdict == "GREEN" or .verdict.verdict == "RED") and ([.files[] | .total == (.passed + .failed + .manual_accepted) and (.total == (.executed + .reused)) and ([.total,.passed,.failed,.manual_accepted,.executed,.reused,.attempts,.flaky] | all(. >= 0 and (floor == .))) ] | all) and @@ -461,100 +477,68 @@ jobs: all(. as $key | ([$summary.files[] | .[$key]] | add // 0) == $summary.totals[$key])) ' "$VERIFIED" >/dev/null 2>&1; then VERIFIED="" - echo 'Verified collector summary unavailable; manual acceptance is unavailable/unverified.' + echo 'Verified report summary unavailable; no verdict, counts or manual acceptance can be shown.' fi + HEADLINE='(no verified report headline)' + FAILURES="" if [ -n "$VERIFIED" ]; then while IFS=$'\t' read -r f T P F M FL EX RE _ATTEMPTS C TIER SHARD; do [ "$T" -eq 0 ] && continue TOTAL=$((TOTAL + T)); PASSED=$((PASSED + P)); FAILED=$((FAILED + F)) - MANUAL=$((MANUAL + M)); FLAKY=$((FLAKY + FL)) + MANUAL=$((MANUAL + M)) EXECUTED=$((EXECUTED + EX)); REUSED=$((REUSED + RE)) COST=$(echo "$COST + $C" | bc) STATUS_ICON="✅" [ "$M" -gt 0 ] && STATUS_ICON="⚠ manual/unscored" [ "$F" -gt 0 ] && STATUS_ICON="❌" - [ "$F" -eq 0 ] && [ "$M" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠" SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | ${M} | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n" done < <(jq -r '.files[] | [.file,.total,.passed,.failed,.manual_accepted,.flaky,.executed,.reused,.attempts,.cost,.tier,.shard] | @tsv' "$VERIFIED") - else - for f in $RESULTS; do - if ! jq -e '.total_tests' "$f" >/dev/null 2>&1; then - echo "Skipping malformed JSON: $f" - continue - fi - # FINAL-attempt accounting: eval-store keeps EVERY retry attempt - # as its own record (that's the flake telemetry), so counting raw - # records marks a pass-on-retry as a failure and inflates totals. - # Group by test name and judge the LAST record. Retry metadata - # includes both passing and failing final outcomes; show it separately. - # Guarded: a file with total_tests but a null/non-array `tests` - # passes the -e probe, the group_by then fails, and an empty $T - # would abort the whole step under bash -e ([ "" -eq 0 ] is an - # error) — killing the comment on exactly the corrupted-artifact - # runs where the red evidence matters (claude adversarial). - STATS=$(jq -r '[.tests | group_by(.name)[] | last] as $final | "\($final | length) \([$final[] | select(.passed)] | length) \([$final[] | select(.passed | not)] | length) \(.flaky_retries // [] | length) \([$final[] | select(.execution != "reused")] | length) \([$final[] | select(.execution == "reused")] | length)"' "$f" 2>/dev/null) || { echo "Skipping malformed tests[] in: $f"; continue; } - read -r T P F FL EX RE <<< "$STATS" - [ -z "$T" ] && { echo "Skipping malformed tests[] in: $f"; continue; } - C=$(jq -r '.total_cost_usd // 0' "$f") - TIER=$(jq -r '.tier // "unknown"' "$f") - SHARD=$(jq -r '.shard // "-"' "$f") - [ "$T" -eq 0 ] && continue - TOTAL=$((TOTAL + T)) - PASSED=$((PASSED + P)) - FAILED=$((FAILED + F)) - FLAKY=$((FLAKY + FL)) - EXECUTED=$((EXECUTED + EX)) - REUSED=$((REUSED + RE)) - COST=$(echo "$COST + $C" | bc) - STATUS_ICON="✅" - [ "$F" -gt 0 ] && STATUS_ICON="❌" - [ "$F" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠" - SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | unverified | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n" - done + # Report-sanitized lines (no @-mentions, one capped line each), fenced here. + HEADLINE=$(jq -r '.headline[]' "$VERIFIED") + FAILURES=$(jq -r '.failures[]' "$VERIFIED") fi COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.' STATUS="✅ PASS" - if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ]; then STATUS="❌ FAIL"; fi + if [ "${RECONCILE_EXIT:-1}" != "0" ]; then STATUS="❌ FAIL"; fi if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi - if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (manual acceptance unavailable/unverified)'; fi + if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (verified report unavailable)'; fi BODY="## E2E Evals: ${STATUS} - **${PASSED} automated passed / ${TOTAL} final results** | **${FAILED} failed, ${MANUAL} manual accepted (unscored; no score-cache credit)** | **${EXECUTED} executed, ${REUSED} reused** | **\$${COST}** total cost | reconcile exit: ${RECONCILE_EXIT:-missing}$([ "$FLAKY" -gt 0 ] && printf ' | ⚠ %s cases with multiple attempts' "$FLAKY") + \`\`\` + ${HEADLINE} + \`\`\` + + **${EXECUTED} executed, ${REUSED} reused** rule/judge records | **${MANUAL} manual accepted (unscored; no score-cache credit)** | **\$${COST}** rule/judge cost | reconcile exit: ${RECONCILE_EXIT:-missing} ${COVERAGE} +
Rule and judge shards + | Shard | Automated result | Manual/unscored | Executed | Reused | Status | Cost | |-------|------------------|-----------------|----------|--------|--------|------| $(echo -e "$SUITE_LINES") +
Fail-closed reconciliation \`\`\` - $(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null || echo '(no reconciliation output)') + $(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo '(no reconciliation output)') \`\`\`
--- - *Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → duration-packed executors → fail-closed report). Reused results retain their original provenance and expiry.*" + *Sliced lane: planner → duration-packed executors → fail-closed report. Behavior cases run a pre-registered 3-trial panel (PASS at 2/3 with no contract violation); a PASS 2/3 is shown with its failed trial, never as a clean pass. Reused results retain their original provenance and expiry.*" - if [ "$FAILED" -gt 0 ]; then - FAILURES="" - for f in $RESULTS; do - if ! jq -e '.failed' "$f" >/dev/null 2>&1; then continue; fi - if [ -n "$VERIFIED" ]; then - FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false and (has("manual_review") | not))][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error") - else - FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false)][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error") - fi - FAILURES="${FAILURES}${FAILS}\n" - done + if [ -n "$FAILURES" ]; then BODY="${BODY} - ### Failures - $(echo -e "$FAILURES")" + ### Failures and split verdicts + \`\`\` + ${FAILURES} + \`\`\`" fi COMMENT_ID=$(gh api repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/comments \ diff --git a/test/ci-paid-coordination.test.ts b/test/ci-paid-coordination.test.ts index 5601e9087..42bea3025 100644 --- a/test/ci-paid-coordination.test.ts +++ b/test/ci-paid-coordination.test.ts @@ -117,10 +117,11 @@ describe('paid CI coordination stays off the eval image', () => { if (name === 'evals.yml') expect(report.permissions).toEqual({ contents: 'read' }); }); - test(`${name}: failure logs include the hidden spool directory without uploading the rest of the cache`, () => { - const logs = jobs['eval-slices'].steps.find(step => step.with?.name === 'paid-slice-${{ matrix.slice }}-logs'); + test(`${name}: shard logs include the hidden spool directory without uploading the rest of the cache`, () => { + const logs = jobs['eval-slices'].steps.find(step => step.with?.name === 'paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}'); expect(logs?.uses).toStartWith('actions/upload-artifact@'); - expect(logs?.if).toBe('failure()'); + // A failed trial no longer reds its runner; its log is still the evidence. + expect(logs?.if).toBe('always()'); expect(logs?.with?.['include-hidden-files']).toBe(true); expect(String(logs?.with?.path).trim().split('\n')).toEqual([ '/home/runner/.cache/gstack-paid-shard-*.log', diff --git a/test/evals-workflow-wiring.test.ts b/test/evals-workflow-wiring.test.ts index 3b34a2783..485f10080 100644 --- a/test/evals-workflow-wiring.test.ts +++ b/test/evals-workflow-wiring.test.ts @@ -263,3 +263,68 @@ describe('shared setup composites (every paid lane)', () => { } }); }); + +describe('panel verdict surfaces (eval reliability policy)', () => { + type AnyJob = { if?: string; needs?: string[]; permissions?: Record; outputs?: Record; + strategy?: { 'max-parallel': number }; steps: Array }; + const jobsOf = (source: string) => (Bun.YAML.parse(source) as { jobs: Record }).jobs; + + test('planners size the capacity preflight with their executor cap', () => { + for (const [source, executor, manifest] of [[evalsYml, 'eval-slices', '/tmp/paid-plan/manifest.json'], + [periodicYml, 'eval-slices', '/tmp/paid-plan/manifest.json'], [periodicYml, 'gate-census', '/tmp/gate-census-plan/manifest.json']] as const) { + const jobs = jobsOf(source); + const emit = jobs['plan-slices']!.steps.find(step => step.run?.includes(`--emit-plan ${manifest} `))!; + const cap = Number(/--max-parallel (\d+)/.exec(emit.run!)?.[1]); + expect(cap, `${executor}: --max-parallel`).toBe(jobs[executor]!.strategy!['max-parallel']); + } + }); + + test('slice artifacts are attempt-scoped and never merged into one tree', () => { + for (const source of [evalsYml, periodicYml, marathonYml]) { + const jobs = jobsOf(source); + const uploads = Object.values(jobs).flatMap(job => job.steps).filter(step => step.uses?.startsWith('actions/upload-artifact@')) + .map(step => step.with?.name ?? '').filter(name => /slice|census-\$/.test(name)); + expect(uploads.length).toBeGreaterThan(0); + for (const name of uploads) expect(name, name).toContain('-a${{ github.run_attempt }}'); + const downloads = Object.values(jobs).flatMap(job => job.steps).filter(step => step.uses?.startsWith('actions/download-artifact@') && step.with?.pattern); + for (const step of downloads) expect((step.with as Record)['merge-multiple'], step.with!.pattern).toBeUndefined(); + } + }); + + test('the PR comment reads collector-outcomes v2 and never recomputes a verdict', () => { + const comment = evalsYml.slice(evalsYml.indexOf(' slices-comment:')); + expect(comment).toContain('.version == 2'); + expect(comment).toContain("jq -r '.failures[]'"); + expect(comment).toContain('name: report-verdict-a${{ github.run_attempt }}'); + expect(evalsYml).not.toContain('group_by(.name)'); + expect(comment).not.toMatch(/paid-slice-/); + const report = jobsOf(evalsYml)['slices-report']!; + expect(report.steps.some(step => step.run?.includes('scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl'))).toBe(true); + expect(report.steps.some(step => step.with?.name?.startsWith('trial-outcomes-'))).toBe(true); + }); + + test('the weekly report gates on pass-rate history, closes its issue on green, and re-dispatches INFRA-only reds once', () => { + const jobs = jobsOf(periodicYml); + const report = jobs.report!; + expect(report.permissions).toEqual({ contents: 'read', issues: 'write', actions: 'read' }); + const gate = report.steps.find(step => step.id === 'pass-rates')!; + expect(gate.run).toContain('bun run eval:pass-rates --gate --runs 10'); + expect(gate.if).toBe('always()'); + for (const name of ['Upsert tracking issue on failure', 'Fail the workflow when reconciliation failed']) { + expect(report.steps.find(step => step.name === name)!.if).toContain("steps.pass-rates.outputs.exit != '0'"); + } + const upsert = report.steps.find(step => step.name === 'Upsert tracking issue on failure')!; + expect(upsert.run).toContain('report-summary.md'); + expect(report.steps.find(step => step.name === 'Close the tracking issue on a green run')!.run).toContain('gh issue close'); + expect(report.steps.filter(step => step.with?.name?.startsWith('trial-outcomes-')).length).toBe(2); + const redispatch = jobs.redispatch!; + expect([redispatch.needs].flat()).toEqual(['report']); + expect(redispatch.permissions).toEqual({ actions: 'write' }); + expect(redispatch.if).toBe("${{ !cancelled() && needs.report.outputs.redispatch == 'true' }}"); + expect(redispatch.steps[0]!.run).toContain('-f redispatch_of="$GITHUB_RUN_ID"'); + const classify = report.steps.find(step => step.id === 'verdict')!; + expect(classify.run).toContain('.verdict.redispatchEligible == true'); + expect(classify.run).toContain('[ -z "$REDISPATCH_OF" ]'); + expect(periodicYml).toMatch(/group: evals-periodic\$\{\{ inputs\.redispatch_of/); + }); +});