mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
567 lines
29 KiB
YAML
567 lines
29 KiB
YAML
name: E2E Evals
|
|
on:
|
|
pull_request:
|
|
branches: [main]
|
|
workflow_dispatch:
|
|
inputs:
|
|
evals_all:
|
|
description: 'Run ALL gate tests in the sliced lane (bypass diff selection; also arms the hollow-shard guard)'
|
|
type: boolean
|
|
default: true
|
|
validation_phase:
|
|
description: 'Validation branch phase; run quality before behavior on unchanged inputs'
|
|
type: choice
|
|
options: [all, quality, cookie-quality, behavior, cookie-behavior]
|
|
default: all
|
|
|
|
concurrency:
|
|
group: evals-${{ github.event.pull_request.number || github.run_id }}
|
|
cancel-in-progress: true
|
|
|
|
env:
|
|
IMAGE: ghcr.io/${{ github.repository }}/ci
|
|
# PRs run changed fast probes; manual runs retain the complete gate census.
|
|
EVALS_PROFILE: ${{ github.event_name == 'pull_request' && 'pr' || 'full' }}
|
|
EVALS_FRESH: ${{ github.event_name == 'workflow_dispatch' && '1' || '' }}
|
|
|
|
jobs:
|
|
# Build Docker image with pre-baked toolchain (cached — only rebuilds on Dockerfile/lockfile change)
|
|
build-image:
|
|
# Dependabot-triggered pull_request runs get a read-only GITHUB_TOKEN, so
|
|
# a lockfile bump = new hash = failed ghcr push = permanently red check
|
|
# (EV6, fork port wave 2). Skip the build for dependabot; the evals job's
|
|
# explicit actor guard mirrors it because no eval test selects on a lockfile-only
|
|
# diff — a maintainer's next push rebuilds the image with real perms.
|
|
if: github.actor != 'dependabot[bot]'
|
|
runs-on: ubicloud-standard-8
|
|
timeout-minutes: 15
|
|
permissions:
|
|
contents: read
|
|
packages: write
|
|
outputs:
|
|
image-tag: ${{ steps.meta.outputs.tag }}
|
|
runtime-id: ${{ steps.runtime.outputs.id }}
|
|
steps:
|
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
|
|
|
- id: meta
|
|
# Key on Dockerfile + lockfile only. package.json is deliberately NOT
|
|
# hashed: its version field changes on every ship (60/60 recent commits),
|
|
# which rebuilt the image each time for a dependency set that only
|
|
# bun.lock determines. A stale baked package.json is harmless — checkout
|
|
# overwrites /workspace and node_modules comes from the lockfile.
|
|
run: echo "tag=${{ env.IMAGE }}:${{ hashFiles('.github/docker/Dockerfile.ci', 'bun.lock', 'patches/**') }}" >> "$GITHUB_OUTPUT"
|
|
|
|
- uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4
|
|
with:
|
|
registry: ghcr.io
|
|
username: ${{ github.actor }}
|
|
password: ${{ secrets.GITHUB_TOKEN }}
|
|
|
|
- name: Check if image exists
|
|
id: check
|
|
run: |
|
|
if docker manifest inspect ${{ steps.meta.outputs.tag }} > /dev/null 2>&1; then
|
|
echo "exists=true" >> "$GITHUB_OUTPUT"
|
|
else
|
|
echo "exists=false" >> "$GITHUB_OUTPUT"
|
|
fi
|
|
|
|
- if: steps.check.outputs.exists == 'false'
|
|
run: cp package.json bun.lock .github/docker/ && cp -R patches .github/docker/patches
|
|
|
|
# A fork PR's GITHUB_TOKEN only has `packages: read`, so pushing fails.
|
|
# Still BUILD (validates Dockerfile.ci changes), just don't publish. This
|
|
# job intentionally keeps no `if:` so fork PRs still get one real, honest
|
|
# green check here instead of a run where every job is grey.
|
|
# Registry cache export needs a docker-container builder — the default
|
|
# `docker` driver hard-errors on cache-to (first live run of the trio).
|
|
- if: steps.check.outputs.exists == 'false'
|
|
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4
|
|
|
|
- if: steps.check.outputs.exists == 'false'
|
|
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7
|
|
with:
|
|
context: .github/docker
|
|
file: .github/docker/Dockerfile.ci
|
|
push: ${{ github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository }}
|
|
# Registry layer cache: reads are safe everywhere; the export is gated
|
|
# to same-repo runs because a fork PR's token can't write GHCR.
|
|
cache-from: type=registry,ref=${{ env.IMAGE }}:buildcache
|
|
cache-to: ${{ (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && format('type=registry,ref={0}:buildcache,mode=max', env.IMAGE) || '' }}
|
|
tags: |
|
|
${{ steps.meta.outputs.tag }}
|
|
${{ env.IMAGE }}:latest
|
|
|
|
- name: Identify the installed eval runtime
|
|
id: runtime
|
|
if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository
|
|
env:
|
|
EVAL_IMAGE: ${{ steps.meta.outputs.tag }}
|
|
run: |
|
|
docker manifest inspect "$EVAL_IMAGE" > /tmp/eval-runtime-manifest.json
|
|
echo "id=$(sha256sum /tmp/eval-runtime-manifest.json | cut -d ' ' -f1)" >> "$GITHUB_OUTPUT"
|
|
|
|
# ── Sliced lane (the ONLY paid lane; legacy 17-row matrix deleted) ──────────
|
|
# One PLANNER computes diff selection + the slice plan ONCE (killing
|
|
# per-slice selector divergence); K executors consume the manifest; the
|
|
# report reconciles results against it FAIL-CLOSED (a slice whose artifact
|
|
# never landed is a failure, a planned shard nobody reported is a failure —
|
|
# hollow lanes cannot aggregate green). Engine: scripts/test-paid-shards.ts —
|
|
# the same runner local eval:bg:gate uses, so CI and local share one
|
|
# selection engine, and every gate-tier file is in the census by
|
|
# construction (no hand-enumerated rows to drift). The legacy matrix ran
|
|
# 18 enumerated files for 22.6 min/$21 per PR serialized AHEAD of this
|
|
# lane's 49-file diff-selected census; parity was demonstrated (sliced
|
|
# census ⊇ matrix files) and the matrix deleted — one revert restores it.
|
|
#
|
|
# Fork PRs never receive repository secrets (ANTHROPIC_API_KEY et al), so
|
|
# every API-calling eval fails at SDK auth before a model runs. Skip
|
|
# deterministically; fork work gets real coverage via a trusted base-repo
|
|
# branch (see CLAUDE.md's garrytan-agents workflow).
|
|
plan-slices:
|
|
runs-on: ubicloud-standard-8
|
|
if: github.actor != 'dependabot[bot]' && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository)
|
|
timeout-minutes: 10
|
|
permissions:
|
|
contents: read
|
|
outputs:
|
|
slices: ${{ steps.matrix.outputs.slices }}
|
|
timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }}
|
|
steps:
|
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
|
with:
|
|
# Preserve full history for merge-base diff selection. Moving the
|
|
# planner off the eval image must not change its selection inputs.
|
|
fetch-depth: 0
|
|
persist-credentials: false
|
|
|
|
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
|
|
with:
|
|
bun-version: 1.4.0
|
|
|
|
# Planner-side reuse: restore this PR's newest receipt store (the report
|
|
# job saves one merged store per run) and ship ONE filtered set with the
|
|
# plan, so every trial of a panel sees the same receipts and a newer FAIL
|
|
# blocks any older PASS for the same inputs.
|
|
- name: Restore this PR's verified judge and E2E results
|
|
if: github.event_name == 'pull_request'
|
|
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
|
|
with:
|
|
path: /tmp/gstack-eval-input-cache
|
|
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-plan
|
|
restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-
|
|
|
|
- name: Emit run manifest
|
|
if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all'
|
|
env:
|
|
EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }}
|
|
EVALS_CACHE_DIR: ${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }}
|
|
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 16
|
|
|
|
- name: Emit validation-phase manifest
|
|
if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all'
|
|
env:
|
|
VALIDATION_PHASE: ${{ inputs.validation_phase }}
|
|
EVALS_ALL: ${{ inputs.evals_all && '1' || '' }}
|
|
EVALS_TIER: gate
|
|
run: |
|
|
bun --no-install -e '
|
|
import { mkdirSync, writeFileSync } from "node:fs";
|
|
import { buildRunManifest, collectPaidTestFiles, restrictManifestSelection } from "./scripts/test-paid-shards.ts";
|
|
const phase = process.env.VALIDATION_PHASE;
|
|
if (!["quality", "cookie-quality", "behavior", "cookie-behavior"].includes(phase)) throw new Error("Invalid validation phase");
|
|
const cookieBehavior = phase === "cookie-behavior";
|
|
const discovered = phase === "cookie-quality" ? ["test/skill-llm-eval.test.ts"]
|
|
: cookieBehavior ? ["test/skill-e2e-bws.test.ts", "test/skill-e2e-qa-workflow.test.ts", "test/skill-e2e-design.test.ts", "test/skill-e2e-diagram.test.ts", "test/skill-e2e-deploy.test.ts"]
|
|
: collectPaidTestFiles().filter(file => file.startsWith("test/skill-llm-eval") === (phase === "quality"));
|
|
const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceBudgetMs: 540000, jobs: 2, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered,
|
|
...(cookieBehavior ? { changedFiles: ["browse/src/cookie-picker-routes.ts", "browse/src/cookie-import-browser.ts", "browse/src/bun-polyfill.cjs"], env: { ...process.env, EVALS_ALL: "" } } : {}) });
|
|
const subset = phase === "cookie-quality" ? { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] }
|
|
: cookieBehavior ? { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] } : null;
|
|
const restricted = subset ? restrictManifestSelection(manifest, subset, "outside the " + phase + " validation subset") : manifest;
|
|
restricted.selectionReason = phase + " validation subset; " + manifest.selectionReason;
|
|
mkdirSync("/tmp/paid-plan", { recursive: true });
|
|
writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(restricted, null, 2) + "\n");
|
|
console.log(phase + ": " + restricted.entries.filter(entry => entry.status === "planned").length + " planned shards");
|
|
'
|
|
|
|
- name: Derive the executor matrix from the plan
|
|
id: matrix
|
|
run: |
|
|
echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
|
echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
|
|
|
|
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
with:
|
|
name: paid-plan
|
|
path: |
|
|
/tmp/paid-plan/manifest.json
|
|
/tmp/paid-plan/receipts
|
|
retention-days: 30
|
|
|
|
eval-slices:
|
|
runs-on: ubicloud-standard-8
|
|
needs: [build-image, plan-slices]
|
|
env:
|
|
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }}
|
|
# !cancelled(), not always(): still runs when build-image was skipped
|
|
# (image already published), but a newer push's cancel-in-progress stops
|
|
# it instead of letting a superseded run finish its paid slices first.
|
|
if: ${{ !cancelled() && needs.build-image.result == 'success' && needs.plan-slices.result == 'success' }}
|
|
# The planner packs ~9 minutes of recorded work per slice (EVALS_JOBS=2 x
|
|
# EVALS_CONCURRENCY=2 per runner, never the old 40-way per-row fan-out
|
|
# that queued claude session STARTUP behind 39 siblings). The job timeout
|
|
# is the plan's supervised worst case plus 20 minutes setup/upload.
|
|
timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }}
|
|
permissions:
|
|
contents: read
|
|
packages: read
|
|
container:
|
|
image: ${{ needs.build-image.outputs.image-tag }}
|
|
credentials:
|
|
username: ${{ github.actor }}
|
|
password: ${{ secrets.GITHUB_TOKEN }}
|
|
options: --user runner
|
|
strategy:
|
|
fail-fast: false
|
|
max-parallel: 16
|
|
matrix:
|
|
slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }}
|
|
steps:
|
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
|
with:
|
|
# Full history: files with SELF-derived selection (the LLM-judge
|
|
# map, routing) walk git at module load, and selection is
|
|
# fail-closed on git errors — a shallow checkout crashed those
|
|
# shards on the lane's first live run ("ambiguous argument
|
|
# 'main...HEAD'"). The manifest still governs WHICH shards run.
|
|
fetch-depth: 0
|
|
persist-credentials: false
|
|
|
|
- name: Fix bun temp
|
|
uses: ./.github/actions/fix-bun-temp
|
|
|
|
- name: Restore deps
|
|
uses: ./.github/actions/restore-deps
|
|
|
|
- run: bun run build
|
|
|
|
# Any slice can host a PTY smoke, so the seed/registration steps run
|
|
# UNCONDITIONALLY (both are idempotent) — the old matrix keyed them on
|
|
# matrix.suite.name, which a sliced lane cannot do. The register
|
|
# composite carries the fail-fast dangling-symlink/frontmatter
|
|
# verification loop, so a moved skill target fails HERE in seconds,
|
|
# not as a wedged PTY session at the shard wall.
|
|
- name: Seed claude interactive config
|
|
uses: ./.github/actions/seed-claude-config
|
|
with:
|
|
anthropic-api-key: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
|
|
- name: Register gstack skills for PTY smokes
|
|
uses: ./.github/actions/register-gstack-skills
|
|
|
|
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
|
with:
|
|
name: paid-plan
|
|
path: /tmp/paid-plan
|
|
|
|
# Receipts come only from the plan (this PR's store, filtered once by the
|
|
# planner); new receipts land beside the slice results and the report
|
|
# merges them into the next store.
|
|
- name: Seed this slice's receipts from the plan
|
|
run: |
|
|
mkdir -p /tmp/paid-slice-results/receipts
|
|
if [ -d /tmp/paid-plan/receipts ]; then cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/; fi
|
|
|
|
- name: Run slice ${{ matrix.slice }}
|
|
env:
|
|
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
|
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
|
PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers
|
|
EVALS_JOBS: "2"
|
|
EVALS_CONCURRENCY: "2"
|
|
GSTACK_EVAL_DIR: /tmp/paid-slice-results
|
|
EVALS_CACHE_DIR: /tmp/paid-slice-results/receipts
|
|
EVALS_CACHE_REPOSITORY: ${{ github.repository }}
|
|
EVALS_CACHE_PR: ${{ github.event.pull_request.number }}
|
|
EVALS_CACHE_RUNTIME_ID: ${{ needs.build-image.outputs.runtime-id }}
|
|
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/paid-plan/manifest.json --slice ${{ matrix.slice }}
|
|
|
|
# Attempt-scoped: a re-run attempt's trials are reported under that
|
|
# attempt and never replace (or collide with) the first attempt's.
|
|
- name: Upload slice results
|
|
if: always()
|
|
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
with:
|
|
name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
|
|
path: /tmp/paid-slice-results
|
|
retention-days: 90
|
|
|
|
- name: Upload native capture evidence
|
|
if: always()
|
|
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
with:
|
|
name: native-captures-${{ env.EVALS_RUN_ID }}
|
|
include-hidden-files: true
|
|
path: |
|
|
~/.gstack/projects/*/e2e-runs
|
|
~/.gstack/projects/*/evals/qa-callers
|
|
~/.gstack-dev/e2e-runs
|
|
~/.gstack-dev/evals/qa-callers
|
|
if-no-files-found: ignore
|
|
retention-days: 90
|
|
|
|
# The spooled per-shard full logs — a red weekly/PR lane three weeks
|
|
# later needs more than a summary line.
|
|
# always(), not failure(): a failed behavior trial is a verdict and no
|
|
# longer reds its runner, but its full log is the diagnosis evidence.
|
|
- name: Upload shard logs
|
|
if: always()
|
|
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
with:
|
|
name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
|
|
include-hidden-files: true
|
|
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
|
|
# runner's spool lands THERE, not /tmp — the original /tmp glob
|
|
# uploaded nothing and a red slice's diagnostics were unreachable.
|
|
path: |
|
|
/home/runner/.cache/gstack-paid-shard-*.log
|
|
/tmp/gstack-paid-shard-*.log
|
|
if-no-files-found: ignore
|
|
retention-days: 30
|
|
|
|
slices-report:
|
|
runs-on: ubicloud-standard-2
|
|
needs: [plan-slices, eval-slices]
|
|
# !cancelled(): the report must run (and FAIL) when an executor died — a
|
|
# missing slice artifact reading as green is the class this lane kills —
|
|
# but a run superseded by a newer push stops here.
|
|
if: ${{ !cancelled() && needs.plan-slices.result == 'success' }}
|
|
timeout-minutes: 5
|
|
# contents:read ONLY — this job executes the PR-authored reconcile
|
|
# runner from the PR checkout, so it
|
|
# must never hold a write-scoped token. The PR comment lives in the
|
|
# separate slices-comment job below, which runs NO repo code: a
|
|
# $GITHUB_ENV/BASH_ENV persistence trick is job-scoped, so the split is
|
|
# the trust boundary (codex adversarial finding, 2026-08-31 — the old
|
|
# matrix-era report job had this separation and the consolidation had
|
|
# regressed it).
|
|
permissions:
|
|
contents: read
|
|
outputs:
|
|
reconcile-exit: ${{ steps.reconcile.outputs.exit }}
|
|
steps:
|
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
|
with:
|
|
persist-credentials: false
|
|
|
|
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
|
|
with:
|
|
bun-version: 1.4.0
|
|
|
|
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
|
with:
|
|
name: paid-plan
|
|
path: /tmp/paid-report
|
|
|
|
# One directory per attempt-scoped slice artifact (no merge): shard
|
|
# records can never overwrite each other, and the report keeps the
|
|
# first attempt's verdict.
|
|
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
|
with:
|
|
pattern: paid-slice-*
|
|
path: /tmp/paid-report
|
|
|
|
- name: Reconcile slices against the manifest (fail-closed)
|
|
id: reconcile
|
|
run: |
|
|
set +e
|
|
EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --report /tmp/paid-report | tee /tmp/report.txt
|
|
# PIPESTATUS[0], NOT $?: GitHub's default run-step shell is
|
|
# `bash -e {0}` with NO pipefail, so $? after the pipe is tee's
|
|
# exit (always 0) — the fail-closed gate was silently fail-open
|
|
# (caught by the ship review army; the wiring test now pins this).
|
|
echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
|
|
|
|
- name: Stamp trial history series
|
|
if: always()
|
|
run: |
|
|
if [ -f /tmp/paid-report/trial-outcomes.jsonl ]; then
|
|
bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl
|
|
fi
|
|
|
|
# One merged receipt store per run: the plan's shipped set, every slice's
|
|
# new pass receipts, and the report's panel and negative receipts. Saved
|
|
# last, so the next planner restores it as the newest prefix match.
|
|
- name: Merge this run's receipts
|
|
if: always() && github.event_name == 'pull_request'
|
|
run: |
|
|
bun --no-install run scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache \
|
|
/tmp/paid-report/receipts /tmp/paid-report/report-receipts /tmp/paid-report/paid-slice-*/receipts
|
|
|
|
- name: Save this PR's verified judge and E2E results
|
|
if: always() && github.event_name == 'pull_request'
|
|
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
|
|
with:
|
|
path: /tmp/gstack-eval-input-cache
|
|
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged
|
|
|
|
- name: Upload reconciliation output for the comment job
|
|
if: always()
|
|
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
with:
|
|
name: report-verdict-a${{ github.run_attempt }}
|
|
path: |
|
|
/tmp/report.txt
|
|
/tmp/paid-report/collector-outcomes.json
|
|
/tmp/paid-report/report-summary.md
|
|
if-no-files-found: ignore
|
|
retention-days: 30
|
|
|
|
- name: Upload trial outcomes for pass-rate history
|
|
if: always()
|
|
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
with:
|
|
name: trial-outcomes-pr-a${{ github.run_attempt }}
|
|
path: /tmp/paid-report/trial-outcomes.jsonl
|
|
if-no-files-found: ignore
|
|
retention-days: 90
|
|
|
|
- name: Fail the workflow when reconciliation failed
|
|
if: steps.reconcile.outputs.exit != '0'
|
|
run: exit 1
|
|
|
|
# PR comment in its OWN job with the write token and ZERO repo code: no
|
|
# checkout, no bun install — only downloaded artifacts, jq, and gh. See the
|
|
# trust-boundary note on slices-report.
|
|
slices-comment:
|
|
runs-on: ubicloud-standard-2
|
|
needs: slices-report
|
|
if: ${{ !cancelled() && github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && needs.slices-report.result != 'skipped' }}
|
|
timeout-minutes: 5
|
|
permissions:
|
|
pull-requests: write
|
|
# The comment upsert calls the REST `/issues/{n}/comments` endpoints
|
|
# (gh api ... issues/comments). With GITHUB_TOKEN those are gated by the
|
|
# `issues` permission, not `pull-requests` (#1802 CI fix).
|
|
issues: write
|
|
steps:
|
|
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
|
with:
|
|
name: paid-plan
|
|
path: /tmp/paid-report
|
|
|
|
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
|
|
with:
|
|
name: report-verdict-a${{ github.run_attempt }}
|
|
path: /tmp/verdict
|
|
continue-on-error: true
|
|
|
|
# Every count, verdict and failure line comes from the read-only report
|
|
# job's collector-outcomes v2 (panelVerdict() ran there); this job runs
|
|
# no repo code and never recomputes a verdict. Keeps the
|
|
# "## E2E Evals" marker so the upsert keeps updating the same comment.
|
|
# Runs even when reconciliation failed — a red lane on the PR is the point.
|
|
- name: Post PR comment
|
|
env:
|
|
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
|
RECONCILE_EXIT: ${{ needs.slices-report.outputs.reconcile-exit }}
|
|
run: |
|
|
# shellcheck disable=SC2086,SC2059
|
|
TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; EXECUTED=0; REUSED=0; COST="0"
|
|
SUITE_LINES=""
|
|
VERIFIED=/tmp/verdict/paid-report/collector-outcomes.json
|
|
if ! jq -e '
|
|
. as $summary |
|
|
.version == 2 and (.files | type == "array") and (.totals | type == "object") and
|
|
(.headline | type == "array") and (.failures | type == "array") and (.panels | type == "array") and
|
|
(.verdict.verdict == "GREEN" or .verdict.verdict == "RED") and
|
|
([.files[] | .total == (.passed + .failed + .manual_accepted) and
|
|
(.total == (.executed + .reused)) and
|
|
([.total,.passed,.failed,.manual_accepted,.executed,.reused,.attempts,.flaky] | all(. >= 0 and (floor == .))) ] | all) and
|
|
(.totals | .total == (.passed + .failed + .manual_accepted) and .total == (.executed + .reused)) and
|
|
(["total","passed","failed","manual_accepted","executed","reused","attempts","flaky"] |
|
|
all(. as $key | ([$summary.files[] | .[$key]] | add // 0) == $summary.totals[$key]))
|
|
' "$VERIFIED" >/dev/null 2>&1; then
|
|
VERIFIED=""
|
|
echo 'Verified report summary unavailable; no verdict, counts or manual acceptance can be shown.'
|
|
fi
|
|
HEADLINE='(no verified report headline)'
|
|
FAILURES=""
|
|
if [ -n "$VERIFIED" ]; then
|
|
while IFS=$'\t' read -r _FILE T P F M _FLAKY EX RE _ATTEMPTS C TIER SHARD; do
|
|
[ "$T" -eq 0 ] && continue
|
|
TOTAL=$((TOTAL + T)); PASSED=$((PASSED + P)); FAILED=$((FAILED + F))
|
|
MANUAL=$((MANUAL + M))
|
|
EXECUTED=$((EXECUTED + EX)); REUSED=$((REUSED + RE))
|
|
COST=$(echo "$COST + $C" | bc)
|
|
STATUS_ICON="✅"
|
|
[ "$M" -gt 0 ] && STATUS_ICON="⚠ manual/unscored"
|
|
[ "$F" -gt 0 ] && STATUS_ICON="❌"
|
|
SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | ${M} | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n"
|
|
done < <(jq -r '.files[] | [.file,.total,.passed,.failed,.manual_accepted,.flaky,.executed,.reused,.attempts,.cost,.tier,.shard] | @tsv' "$VERIFIED")
|
|
# Report-sanitized lines (no @-mentions, one capped line each), fenced here.
|
|
HEADLINE=$(jq -r '.headline[]' "$VERIFIED")
|
|
FAILURES=$(jq -r '.failures[]' "$VERIFIED")
|
|
fi
|
|
|
|
COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.'
|
|
|
|
STATUS="✅ PASS"
|
|
if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ] \
|
|
|| { [ -n "$VERIFIED" ] && [ "$(jq -r '.verdict.verdict' "$VERIFIED")" != "GREEN" ]; }; then STATUS="❌ FAIL"; fi
|
|
if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi
|
|
if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (verified report unavailable)'; fi
|
|
|
|
BODY="## E2E Evals: ${STATUS}
|
|
|
|
\`\`\`
|
|
${HEADLINE}
|
|
\`\`\`
|
|
|
|
**${EXECUTED} executed, ${REUSED} reused** rule/judge records | **${MANUAL} manual accepted (unscored; no score-cache credit)** | **\$${COST}** rule/judge cost | reconcile exit: ${RECONCILE_EXIT:-missing}
|
|
|
|
${COVERAGE}
|
|
|
|
<details><summary>Rule and judge shards</summary>
|
|
|
|
| Shard | Automated result | Manual/unscored | Executed | Reused | Status | Cost |
|
|
|-------|------------------|-----------------|----------|--------|--------|------|
|
|
$(echo -e "$SUITE_LINES")
|
|
</details>
|
|
|
|
<details><summary>Fail-closed reconciliation</summary>
|
|
|
|
\`\`\`
|
|
$(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo '(no reconciliation output)')
|
|
\`\`\`
|
|
</details>
|
|
|
|
---
|
|
*Sliced lane: planner → duration-packed executors → fail-closed report. Behavior cases run a pre-registered 3-trial panel (PASS at 2/3 with no contract violation); a PASS 2/3 is shown with its failed trial, never as a clean pass. Reused results retain their original provenance and expiry.*"
|
|
|
|
if [ -n "$FAILURES" ]; then
|
|
BODY="${BODY}
|
|
|
|
### Failures and split verdicts
|
|
\`\`\`
|
|
${FAILURES}
|
|
\`\`\`"
|
|
fi
|
|
|
|
COMMENT_ID=$(gh api repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/comments \
|
|
--jq '.[] | select(.body | startswith("## E2E Evals")) | .id' | tail -1)
|
|
|
|
if [ -n "$COMMENT_ID" ]; then
|
|
gh api "repos/${{ github.repository }}/issues/comments/${COMMENT_ID}" \
|
|
-X PATCH -f body="$BODY"
|
|
else
|
|
# REST, not gh's pr-comment subcommand: this job runs with NO
|
|
# checkout (the token/exec split), and that subcommand resolves
|
|
# the repo FROM git — it dies with "not a git repository" here.
|
|
gh api "repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/comments" \
|
|
-X POST -f body="$BODY"
|
|
fi
|