mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
ci(evals): attempt-scoped artifacts, verdict-v2 PR comment, weekly pass-rate gate and one INFRA re-dispatch
- Slice, census and marathon artifacts carry -a<run_attempt>; reports download them per artifact (no merge), so records never overwrite and a re-run never replaces the first attempt's verdict. - Planners pass --max-parallel for the capacity preflight (24/16 unchanged: the refreshed periodic plan needs 24 slices, the gate census 12). - PR comment: jq-only job reads collector-outcomes v2 (headline, sanitized failure block); the group_by(.name)|last recomputation is gone. - Reports stamp series identities, upload trial-outcomes-* for history, and shard logs upload always (a failed trial no longer reds its runner). - Weekly report: headline + failure block of both lanes in the issue body, the eval:pass-rates --gate step (fails closed without history), close the issue on a green run, and UC-E1: when every red is machine-classified INFRA/INCOMPLETE, one re-dispatch as a new run in its own concurrency group (redispatch_of), both runs reported.
This commit is contained in:
1 parent
62fb9a255d
commit
8622535b90
5 files changed
+284
-111
No files matched your search
+60
-76
@@ -144,7 +144,7 @@ jobs:
|
||||
if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all'
|
||||
env:
|
||||
EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }}
|
||||
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2
|
||||
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 16
|
||||
|
||||
- name: Emit validation-phase manifest
|
||||
if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all'
|
||||
@@ -299,11 +299,13 @@ jobs:
|
||||
path: /tmp/gstack-eval-input-cache
|
||||
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}
|
||||
|
||||
# Attempt-scoped: a re-run attempt's trials are reported under that
|
||||
# attempt and never replace (or collide with) the first attempt's.
|
||||
- name: Upload slice results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: paid-slice-${{ matrix.slice }}
|
||||
name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
|
||||
path: /tmp/paid-slice-results
|
||||
retention-days: 90
|
||||
|
||||
@@ -323,11 +325,13 @@ jobs:
|
||||
|
||||
# The spooled per-shard full logs — a red weekly/PR lane three weeks
|
||||
# later needs more than a summary line.
|
||||
- name: Upload shard logs on failure
|
||||
if: failure()
|
||||
# always(), not failure(): a failed behavior trial is a verdict and no
|
||||
# longer reds its runner, but its full log is the diagnosis evidence.
|
||||
- name: Upload shard logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: paid-slice-${{ matrix.slice }}-logs
|
||||
name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
|
||||
include-hidden-files: true
|
||||
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
|
||||
# runner's spool lands THERE, not /tmp — the original /tmp glob
|
||||
@@ -372,11 +376,13 @@ jobs:
|
||||
name: paid-plan
|
||||
path: /tmp/paid-report
|
||||
|
||||
# One directory per attempt-scoped slice artifact (no merge): shard
|
||||
# records can never overwrite each other, and the report keeps the
|
||||
# first attempt's verdict.
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
||||
with:
|
||||
pattern: paid-slice-[0-9]*
|
||||
pattern: paid-slice-*
|
||||
path: /tmp/paid-report
|
||||
merge-multiple: true
|
||||
|
||||
- name: Reconcile slices against the manifest (fail-closed)
|
||||
id: reconcile
|
||||
@@ -389,17 +395,31 @@ jobs:
|
||||
# (caught by the ship review army; the wiring test now pins this).
|
||||
echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Stamp trial history series
|
||||
if: always() && hashFiles('/tmp/paid-report/trial-outcomes.jsonl') != ''
|
||||
run: bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl
|
||||
|
||||
- name: Upload reconciliation output for the comment job
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: report-verdict
|
||||
name: report-verdict-a${{ github.run_attempt }}
|
||||
path: |
|
||||
/tmp/report.txt
|
||||
/tmp/paid-report/collector-outcomes.json
|
||||
/tmp/paid-report/report-summary.md
|
||||
if-no-files-found: ignore
|
||||
retention-days: 30
|
||||
|
||||
- name: Upload trial outcomes for pass-rate history
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: trial-outcomes-pr-a${{ github.run_attempt }}
|
||||
path: /tmp/paid-report/trial-outcomes.jsonl
|
||||
if-no-files-found: ignore
|
||||
retention-days: 90
|
||||
|
||||
- name: Fail the workflow when reconciliation failed
|
||||
if: steps.reconcile.outputs.exit != '0'
|
||||
run: exit 1
|
||||
@@ -426,18 +446,13 @@ jobs:
|
||||
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
||||
with:
|
||||
pattern: paid-slice-[0-9]*
|
||||
path: /tmp/paid-report
|
||||
merge-multiple: true
|
||||
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
||||
with:
|
||||
name: report-verdict
|
||||
name: report-verdict-a${{ github.run_attempt }}
|
||||
path: /tmp/verdict
|
||||
continue-on-error: true
|
||||
|
||||
# Verified counts come from the read-only report job, not repo code in
|
||||
# this write-token job. Keeps the
|
||||
# Every count, verdict and failure line comes from the read-only report
|
||||
# job's collector-outcomes v2 (panelVerdict() ran there); this job runs
|
||||
# no repo code and never recomputes a verdict. Keeps the
|
||||
# "## E2E Evals" marker so the upsert keeps updating the same comment.
|
||||
# Runs even when reconciliation failed — a red lane on the PR is the point.
|
||||
- name: Post PR comment
|
||||
@@ -446,13 +461,14 @@ jobs:
|
||||
RECONCILE_EXIT: ${{ needs.slices-report.outputs.reconcile-exit }}
|
||||
run: |
|
||||
# shellcheck disable=SC2086,SC2059
|
||||
RESULTS=$(find /tmp/paid-report -name '*.json' ! -name 'manifest.json' ! -name 'slice-*.json' ! -name '_partial*' 2>/dev/null | sort)
|
||||
TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; FLAKY=0; EXECUTED=0; REUSED=0; COST="0"
|
||||
TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; EXECUTED=0; REUSED=0; COST="0"
|
||||
SUITE_LINES=""
|
||||
VERIFIED=/tmp/verdict/paid-report/collector-outcomes.json
|
||||
if ! jq -e '
|
||||
. as $summary |
|
||||
.version == 1 and (.files | type == "array") and (.totals | type == "object") and
|
||||
.version == 2 and (.files | type == "array") and (.totals | type == "object") and
|
||||
(.headline | type == "array") and (.failures | type == "array") and (.panels | type == "array") and
|
||||
(.verdict.verdict == "GREEN" or .verdict.verdict == "RED") and
|
||||
([.files[] | .total == (.passed + .failed + .manual_accepted) and
|
||||
(.total == (.executed + .reused)) and
|
||||
([.total,.passed,.failed,.manual_accepted,.executed,.reused,.attempts,.flaky] | all(. >= 0 and (floor == .))) ] | all) and
|
||||
@@ -461,100 +477,68 @@ jobs:
|
||||
all(. as $key | ([$summary.files[] | .[$key]] | add // 0) == $summary.totals[$key]))
|
||||
' "$VERIFIED" >/dev/null 2>&1; then
|
||||
VERIFIED=""
|
||||
echo 'Verified collector summary unavailable; manual acceptance is unavailable/unverified.'
|
||||
echo 'Verified report summary unavailable; no verdict, counts or manual acceptance can be shown.'
|
||||
fi
|
||||
HEADLINE='(no verified report headline)'
|
||||
FAILURES=""
|
||||
if [ -n "$VERIFIED" ]; then
|
||||
while IFS=$'\t' read -r f T P F M FL EX RE _ATTEMPTS C TIER SHARD; do
|
||||
[ "$T" -eq 0 ] && continue
|
||||
TOTAL=$((TOTAL + T)); PASSED=$((PASSED + P)); FAILED=$((FAILED + F))
|
||||
MANUAL=$((MANUAL + M)); FLAKY=$((FLAKY + FL))
|
||||
MANUAL=$((MANUAL + M))
|
||||
EXECUTED=$((EXECUTED + EX)); REUSED=$((REUSED + RE))
|
||||
COST=$(echo "$COST + $C" | bc)
|
||||
STATUS_ICON="✅"
|
||||
[ "$M" -gt 0 ] && STATUS_ICON="⚠ manual/unscored"
|
||||
[ "$F" -gt 0 ] && STATUS_ICON="❌"
|
||||
[ "$F" -eq 0 ] && [ "$M" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠"
|
||||
SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | ${M} | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n"
|
||||
done < <(jq -r '.files[] | [.file,.total,.passed,.failed,.manual_accepted,.flaky,.executed,.reused,.attempts,.cost,.tier,.shard] | @tsv' "$VERIFIED")
|
||||
else
|
||||
for f in $RESULTS; do
|
||||
if ! jq -e '.total_tests' "$f" >/dev/null 2>&1; then
|
||||
echo "Skipping malformed JSON: $f"
|
||||
continue
|
||||
fi
|
||||
# FINAL-attempt accounting: eval-store keeps EVERY retry attempt
|
||||
# as its own record (that's the flake telemetry), so counting raw
|
||||
# records marks a pass-on-retry as a failure and inflates totals.
|
||||
# Group by test name and judge the LAST record. Retry metadata
|
||||
# includes both passing and failing final outcomes; show it separately.
|
||||
# Guarded: a file with total_tests but a null/non-array `tests`
|
||||
# passes the -e probe, the group_by then fails, and an empty $T
|
||||
# would abort the whole step under bash -e ([ "" -eq 0 ] is an
|
||||
# error) — killing the comment on exactly the corrupted-artifact
|
||||
# runs where the red evidence matters (claude adversarial).
|
||||
STATS=$(jq -r '[.tests | group_by(.name)[] | last] as $final | "\($final | length) \([$final[] | select(.passed)] | length) \([$final[] | select(.passed | not)] | length) \(.flaky_retries // [] | length) \([$final[] | select(.execution != "reused")] | length) \([$final[] | select(.execution == "reused")] | length)"' "$f" 2>/dev/null) || { echo "Skipping malformed tests[] in: $f"; continue; }
|
||||
read -r T P F FL EX RE <<< "$STATS"
|
||||
[ -z "$T" ] && { echo "Skipping malformed tests[] in: $f"; continue; }
|
||||
C=$(jq -r '.total_cost_usd // 0' "$f")
|
||||
TIER=$(jq -r '.tier // "unknown"' "$f")
|
||||
SHARD=$(jq -r '.shard // "-"' "$f")
|
||||
[ "$T" -eq 0 ] && continue
|
||||
TOTAL=$((TOTAL + T))
|
||||
PASSED=$((PASSED + P))
|
||||
FAILED=$((FAILED + F))
|
||||
FLAKY=$((FLAKY + FL))
|
||||
EXECUTED=$((EXECUTED + EX))
|
||||
REUSED=$((REUSED + RE))
|
||||
COST=$(echo "$COST + $C" | bc)
|
||||
STATUS_ICON="✅"
|
||||
[ "$F" -gt 0 ] && STATUS_ICON="❌"
|
||||
[ "$F" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠"
|
||||
SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | unverified | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n"
|
||||
done
|
||||
# Report-sanitized lines (no @-mentions, one capped line each), fenced here.
|
||||
HEADLINE=$(jq -r '.headline[]' "$VERIFIED")
|
||||
FAILURES=$(jq -r '.failures[]' "$VERIFIED")
|
||||
fi
|
||||
|
||||
COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.'
|
||||
|
||||
STATUS="✅ PASS"
|
||||
if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ]; then STATUS="❌ FAIL"; fi
|
||||
if [ "${RECONCILE_EXIT:-1}" != "0" ]; then STATUS="❌ FAIL"; fi
|
||||
if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi
|
||||
if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (manual acceptance unavailable/unverified)'; fi
|
||||
if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (verified report unavailable)'; fi
|
||||
|
||||
BODY="## E2E Evals: ${STATUS}
|
||||
|
||||
**${PASSED} automated passed / ${TOTAL} final results** | **${FAILED} failed, ${MANUAL} manual accepted (unscored; no score-cache credit)** | **${EXECUTED} executed, ${REUSED} reused** | **\$${COST}** total cost | reconcile exit: ${RECONCILE_EXIT:-missing}$([ "$FLAKY" -gt 0 ] && printf ' | ⚠ %s cases with multiple attempts' "$FLAKY")
|
||||
\`\`\`
|
||||
${HEADLINE}
|
||||
\`\`\`
|
||||
|
||||
**${EXECUTED} executed, ${REUSED} reused** rule/judge records | **${MANUAL} manual accepted (unscored; no score-cache credit)** | **\$${COST}** rule/judge cost | reconcile exit: ${RECONCILE_EXIT:-missing}
|
||||
|
||||
${COVERAGE}
|
||||
|
||||
<details><summary>Rule and judge shards</summary>
|
||||
|
||||
| Shard | Automated result | Manual/unscored | Executed | Reused | Status | Cost |
|
||||
|-------|------------------|-----------------|----------|--------|--------|------|
|
||||
$(echo -e "$SUITE_LINES")
|
||||
</details>
|
||||
|
||||
<details><summary>Fail-closed reconciliation</summary>
|
||||
|
||||
\`\`\`
|
||||
$(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null || echo '(no reconciliation output)')
|
||||
$(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo '(no reconciliation output)')
|
||||
\`\`\`
|
||||
</details>
|
||||
|
||||
---
|
||||
*Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → duration-packed executors → fail-closed report). Reused results retain their original provenance and expiry.*"
|
||||
*Sliced lane: planner → duration-packed executors → fail-closed report. Behavior cases run a pre-registered 3-trial panel (PASS at 2/3 with no contract violation); a PASS 2/3 is shown with its failed trial, never as a clean pass. Reused results retain their original provenance and expiry.*"
|
||||
|
||||
if [ "$FAILED" -gt 0 ]; then
|
||||
FAILURES=""
|
||||
for f in $RESULTS; do
|
||||
if ! jq -e '.failed' "$f" >/dev/null 2>&1; then continue; fi
|
||||
if [ -n "$VERIFIED" ]; then
|
||||
FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false and (has("manual_review") | not))][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error")
|
||||
else
|
||||
FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false)][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error")
|
||||
fi
|
||||
FAILURES="${FAILURES}${FAILS}\n"
|
||||
done
|
||||
if [ -n "$FAILURES" ]; then
|
||||
BODY="${BODY}
|
||||
|
||||
### Failures
|
||||
$(echo -e "$FAILURES")"
|
||||
### Failures and split verdicts
|
||||
\`\`\`
|
||||
${FAILURES}
|
||||
\`\`\`"
|
||||
fi
|
||||
|
||||
COMMENT_ID=$(gh api repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/comments \
|
||||
|
||||
Reference in new issue
Block a user