Merge remote-tracking branch 'origin/capy/rel-a' into capy/rel-c

This commit is contained in:
garrytan committed 2026-09-29 19:47:17 +00:00
commit b2ca207cf0
46 files changed
+4936 -936

No files matched your search

+3 -4
View File
@@ -187,7 +187,7 @@ jobs:
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: marathon-slice-${{ matrix.slice }}
name: marathon-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
path: /tmp/marathon-slice-results
retention-days: 90
@@ -209,7 +209,7 @@ jobs:
if: failure()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: marathon-slice-${{ matrix.slice }}-logs
name: marathon-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
include-hidden-files: true
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
# runner's spool lands THERE, not /tmp — the original /tmp glob
@@ -247,9 +247,8 @@ jobs:
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
pattern: marathon-slice-[0-9]*
pattern: marathon-slice-*
path: /tmp/marathon-report
merge-multiple: true
- name: Reconcile slices against the manifest (fail-closed)
id: reconcile
+152 -28
View File
@@ -19,9 +19,16 @@ on:
schedule:
- cron: '0 6 * * 1' # Monday 6 AM UTC (ci-image prebuilds at 4 AM)
workflow_dispatch:
inputs:
redispatch_of:
description: 'Run id this run re-dispatches (the one INFRA/INCOMPLETE-only re-dispatch; set by the report job)'
type: string
default: ''
# A re-dispatch runs in its own group so it never cancels the run that
# dispatched it; both runs are reported.
concurrency:
group: evals-periodic
group: evals-periodic${{ inputs.redispatch_of && format('-redispatch-{0}', inputs.redispatch_of) || '' }}
cancel-in-progress: true
env:
@@ -106,7 +113,7 @@ jobs:
- name: Emit run manifest (ALL periodic tests minus reasoned excludes)
env:
EVALS_ALL: "1"
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2
run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 24
- name: Derive the periodic executor matrix from the plan
id: periodic-matrix
@@ -123,7 +130,7 @@ jobs:
- name: Emit gate census manifest (ALL gate tests)
env:
EVALS_ALL: "1"
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges --max-parallel 16
- name: Derive the gate census executor matrix from the plan
id: gate-matrix
@@ -214,7 +221,7 @@ jobs:
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: paid-slice-${{ matrix.slice }}
name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
path: /tmp/paid-slice-results
retention-days: 90
@@ -232,11 +239,13 @@ jobs:
if-no-files-found: ignore
retention-days: 90
- name: Upload shard logs on failure
if: failure()
# always(), not failure(): a failed behavior trial is a verdict and no
# longer reds its runner, but its full log is the diagnosis evidence.
- name: Upload shard logs
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: paid-slice-${{ matrix.slice }}-logs
name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
include-hidden-files: true
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
# runner's spool lands THERE, not /tmp — the original /tmp glob
@@ -312,7 +321,7 @@ jobs:
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: gate-census-${{ matrix.slice }}
name: gate-census-${{ matrix.slice }}-a${{ github.run_attempt }}
path: /tmp/gate-census-results
retention-days: 90
@@ -337,12 +346,16 @@ jobs:
# missing slice artifact reading as green is the class this lane kills —
# but a cancelled run stops here.
if: ${{ !cancelled() && needs.plan-slices.result == 'success' }}
timeout-minutes: 10
timeout-minutes: 15
permissions:
contents: read
# The failure notification below upserts a tracking issue via
# `gh api /issues` — gated by the issues permission.
# The notification below upserts (or closes) a tracking issue via
# `gh issue` — gated by the issues permission.
issues: write
# Pass-rate history downloads earlier weekly runs' trial-outcomes.
actions: read
outputs:
redispatch: ${{ steps.verdict.outputs.redispatch }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -357,11 +370,12 @@ jobs:
name: paid-plan
path: /tmp/paid-report
# One directory per attempt-scoped slice artifact (no merge): shard
# records never overwrite each other and the first attempt decides.
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
pattern: paid-slice-[0-9]*
pattern: paid-slice-*
path: /tmp/paid-report
merge-multiple: true
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
@@ -372,7 +386,6 @@ jobs:
with:
pattern: gate-census-[0-9]*
path: /tmp/gate-census-report
merge-multiple: true
- name: Reconcile slices against the manifest (fail-closed)
id: reconcile
@@ -394,34 +407,114 @@ jobs:
EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --report /tmp/gate-census-report | tee /tmp/gate-report.txt
echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
# A red weekly lane nobody must action is waste — upsert ONE tracking
# issue (never a new issue per week) with the reconciliation output, so
# failures have an owner-visible artifact with history in one place.
- name: Upsert tracking issue on failure
if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success')
- name: Stamp trial history series
if: always()
run: |
for file in /tmp/paid-report/trial-outcomes.jsonl /tmp/gate-census-report/trial-outcomes.jsonl; do
if [ -f "$file" ]; then bun --no-install run scripts/eval-trial-series.ts "$file"; fi
done
- name: Upload trial outcomes for pass-rate history
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: trial-outcomes-periodic-a${{ github.run_attempt }}
path: |
/tmp/paid-report/trial-outcomes.jsonl
if-no-files-found: ignore
retention-days: 90
- name: Upload gate census trial outcomes for pass-rate history
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: trial-outcomes-gate-census-a${{ github.run_attempt }}
path: |
/tmp/gate-census-report/trial-outcomes.jsonl
if-no-files-found: ignore
retention-days: 90
# Weekly pass-rate gate over the last 10 weekly runs (drift, rule cases
# behaving like behavior, quarantine exit/expiry/cap). Fails closed when
# history cannot be fetched.
- name: Pass-rate history gate
id: pass-rates
if: always()
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set +e
bun run eval:pass-rates --gate --runs 10 > /tmp/pass-rates.txt 2>&1
echo "exit=$?" >> "$GITHUB_OUTPUT"
cat /tmp/pass-rates.txt
# UC-E1 (approved): a run whose every red verdict is machine-classified
# INFRA or INCOMPLETE may be re-dispatched ONCE as a new run.
- name: Classify the census verdict
id: verdict
if: always()
env:
REDISPATCH_OF: ${{ inputs.redispatch_of }}
PERIODIC_EXIT: ${{ steps.reconcile.outputs.exit }}
GATE_EXIT: ${{ steps.gate-reconcile.outputs.exit }}
run: |
eligible() { # $1 exit, $2 report dir
[ "$1" = "0" ] && return 0
jq -e '.version == 2 and .verdict.redispatchEligible == true' "$2/collector-outcomes.json" >/dev/null 2>&1
}
if [ -z "$REDISPATCH_OF" ] && { [ "$PERIODIC_EXIT" != "0" ] || [ "$GATE_EXIT" != "0" ]; } \
&& eligible "$PERIODIC_EXIT" /tmp/paid-report && eligible "$GATE_EXIT" /tmp/gate-census-report; then
echo "redispatch=true" >> "$GITHUB_OUTPUT"
else
echo "redispatch=false" >> "$GITHUB_OUTPUT"
fi
# A red weekly lane nobody must action is waste — upsert ONE tracking
# issue (never a new issue per week) with the headline and failure block
# of both lanes, and close it on the next green run.
- name: Upsert tracking issue on failure
if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || steps.pass-rates.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success')
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
REDISPATCH: ${{ steps.verdict.outputs.redispatch }}
REDISPATCH_OF: ${{ inputs.redispatch_of }}
run: |
set -euo pipefail
TITLE="Weekly periodic evals: red lane needs triage"
BODY_FILE=/tmp/issue-body.md
RUN_URL="${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
{
echo "Automated weekly report — run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
echo "Automated weekly report — run: ${RUN_URL}"
if [ -n "$REDISPATCH_OF" ]; then echo; echo "This run is the one INFRA/INCOMPLETE re-dispatch of run ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${REDISPATCH_OF}; both runs are reported."; fi
if [ "$REDISPATCH" = "true" ]; then echo; echo "Every red verdict is machine-classified INFRA/INCOMPLETE: re-dispatching once as a new run (EVAL_POLICY.infraRedispatch). This run stays red and reported."; fi
echo
echo "- periodic reconciliation exit: ${{ steps.reconcile.outputs.exit }}"
echo "- periodic slices job: ${{ needs.eval-slices.result }}"
echo "- gate census reconciliation exit: ${{ steps.gate-reconcile.outputs.exit }}"
echo "- gate census job: ${{ needs.gate-census.result }}"
echo "- periodic reconciliation exit: ${{ steps.reconcile.outputs.exit }} (slices job: ${{ needs.eval-slices.result }})"
echo "- gate census reconciliation exit: ${{ steps.gate-reconcile.outputs.exit }} (census job: ${{ needs.gate-census.result }})"
echo "- pass-rate history gate exit: ${{ steps.pass-rates.outputs.exit }}"
echo
echo "### Periodic lane"
cat /tmp/paid-report/report-summary.md 2>/dev/null || echo "(no periodic report summary)"
echo
echo "### Gate census"
cat /tmp/gate-census-report/report-summary.md 2>/dev/null || echo "(no gate census report summary)"
echo
echo "### Pass-rate history (ACTION REQUIRED)"
echo '```'
{ grep -E 'ACTION REQUIRED|history unavailable' /tmp/pass-rates.txt || echo "(no pass-rate alarms)"; } | sed 's/@/@\xe2\x80\x8b/g' | head -c 6000
echo '```'
echo
echo "<details><summary>Full reconciliation output</summary>"
echo
echo '```'
tail -c 6000 /tmp/report.txt 2>/dev/null || echo "(no reconciliation output)"
tail -c 6000 /tmp/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo "(no reconciliation output)"
echo '```'
echo
echo '```'
tail -c 6000 /tmp/gate-report.txt 2>/dev/null || echo "(no gate census reconciliation output)"
tail -c 6000 /tmp/gate-report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo "(no gate census reconciliation output)"
echo '```'
echo "</details>"
echo
echo "Exclusion policy: test/helpers/periodic-exclude-data.ts (every entry needs reason + tracking; removal re-activates the file next week)."
echo "Policy: EVAL_POLICY and CASE_QUARANTINE in test/helpers/periodic-exclude-data.ts; history: \`bun run eval:pass-rates\`."
} > "$BODY_FILE"
EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty')
if [ -n "$EXISTING" ]; then
@@ -431,6 +524,37 @@ jobs:
gh issue create --repo "$GITHUB_REPOSITORY" --title "$TITLE" --body-file "$BODY_FILE"
fi
- name: Close the tracking issue on a green run
if: always() && steps.reconcile.outputs.exit == '0' && steps.gate-reconcile.outputs.exit == '0' && steps.pass-rates.outputs.exit == '0' && needs.eval-slices.result == 'success' && needs.gate-census.result == 'success'
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
REDISPATCH_OF: ${{ inputs.redispatch_of }}
run: |
set -euo pipefail
TITLE="Weekly periodic evals: red lane needs triage"
EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty')
if [ -n "$EXISTING" ]; then
NOTE="Green weekly run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
if [ -n "$REDISPATCH_OF" ]; then NOTE="${NOTE} (the INFRA re-dispatch of run ${REDISPATCH_OF}, which stays red and reported)"; fi
gh issue close "$EXISTING" --repo "$GITHUB_REPOSITORY" --comment "$NOTE"
fi
- name: Fail the workflow when reconciliation failed
if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success')
if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || steps.pass-rates.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success')
run: exit 1
# The one INFRA/INCOMPLETE re-dispatch (UC-E1). Its own job so the report
# job keeps no actions:write; the new run's concurrency group differs, so it
# never cancels this run.
redispatch:
runs-on: ubicloud-standard-2
needs: report
if: ${{ !cancelled() && needs.report.outputs.redispatch == 'true' }}
timeout-minutes: 5
permissions:
actions: write
steps:
- name: Re-dispatch the weekly census once
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: gh workflow run evals-periodic.yml --repo "$GITHUB_REPOSITORY" --ref "$GITHUB_REF_NAME" -f redispatch_of="$GITHUB_RUN_ID"
+104 -110
View File
@@ -140,11 +140,24 @@ jobs:
with:
bun-version: 1.4.0
# Planner-side reuse: restore this PR's newest receipt store (the report
# job saves one merged store per run) and ship ONE filtered set with the
# plan, so every trial of a panel sees the same receipts and a newer FAIL
# blocks any older PASS for the same inputs.
- name: Restore this PR's verified judge and E2E results
if: github.event_name == 'pull_request'
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
with:
path: /tmp/gstack-eval-input-cache
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-plan
restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-
- name: Emit run manifest
if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all'
env:
EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }}
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2
EVALS_CACHE_DIR: ${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }}
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 16
- name: Emit validation-phase manifest
if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all'
@@ -182,7 +195,9 @@ jobs:
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: paid-plan
path: /tmp/paid-plan/manifest.json
path: |
/tmp/paid-plan/manifest.json
/tmp/paid-plan/receipts
retention-days: 30
eval-slices:
@@ -251,15 +266,13 @@ jobs:
name: paid-plan
path: /tmp/paid-plan
# Only this PR's receipts are eligible. No base-branch or cross-PR restore
# prefix; every receipt also verifies exact inputs and its original age.
- name: Restore this PR's verified judge and E2E results
if: github.event_name == 'pull_request'
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
with:
path: /tmp/gstack-eval-input-cache
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}
restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-
# Receipts come only from the plan (this PR's store, filtered once by the
# planner); new receipts land beside the slice results and the report
# merges them into the next store.
- name: Seed this slice's receipts from the plan
run: |
mkdir -p /tmp/paid-slice-results/receipts
if [ -d /tmp/paid-plan/receipts ]; then cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/; fi
- name: Run slice ${{ matrix.slice }}
env:
@@ -270,40 +283,19 @@ jobs:
EVALS_JOBS: "2"
EVALS_CONCURRENCY: "2"
GSTACK_EVAL_DIR: /tmp/paid-slice-results
EVALS_CACHE_DIR: /tmp/gstack-eval-input-cache
EVALS_CACHE_DIR: /tmp/paid-slice-results/receipts
EVALS_CACHE_REPOSITORY: ${{ github.repository }}
EVALS_CACHE_PR: ${{ github.event.pull_request.number }}
EVALS_CACHE_RUNTIME_ID: ${{ needs.build-image.outputs.runtime-id }}
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/paid-plan/manifest.json --slice ${{ matrix.slice }}
- name: Find finalized passing receipts
id: receipts
if: ${{ !cancelled() && github.event_name == 'pull_request' }}
run: |
# Only a producer publishes. A later reuse-only slice must not become
# the newest prefix match and hide another slice's newly earned pass.
for receipt in /tmp/gstack-eval-input-cache/*.json; do
[ -f "$receipt" ] || continue
if jq -e --arg run "$GITHUB_RUN_ID/$GITHUB_RUN_ATTEMPT" '.proof.source.runId == $run' "$receipt" >/dev/null 2>&1; then
echo 'present=true' >> "$GITHUB_OUTPUT"
break
fi
done
# An unrelated failing case does not discard already verified passes.
# Failed/retried/partial attempts never become receipts in the first place.
- name: Save verified judge and E2E results for this PR
if: ${{ !cancelled() && steps.receipts.outputs.present == 'true' }}
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
with:
path: /tmp/gstack-eval-input-cache
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}
# Attempt-scoped: a re-run attempt's trials are reported under that
# attempt and never replace (or collide with) the first attempt's.
- name: Upload slice results
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: paid-slice-${{ matrix.slice }}
name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
path: /tmp/paid-slice-results
retention-days: 90
@@ -323,11 +315,13 @@ jobs:
# The spooled per-shard full logs — a red weekly/PR lane three weeks
# later needs more than a summary line.
- name: Upload shard logs on failure
if: failure()
# always(), not failure(): a failed behavior trial is a verdict and no
# longer reds its runner, but its full log is the diagnosis evidence.
- name: Upload shard logs
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: paid-slice-${{ matrix.slice }}-logs
name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
include-hidden-files: true
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
# runner's spool lands THERE, not /tmp — the original /tmp glob
@@ -372,11 +366,13 @@ jobs:
name: paid-plan
path: /tmp/paid-report
# One directory per attempt-scoped slice artifact (no merge): shard
# records can never overwrite each other, and the report keeps the
# first attempt's verdict.
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
pattern: paid-slice-[0-9]*
pattern: paid-slice-*
path: /tmp/paid-report
merge-multiple: true
- name: Reconcile slices against the manifest (fail-closed)
id: reconcile
@@ -389,17 +385,50 @@ jobs:
# (caught by the ship review army; the wiring test now pins this).
echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
- name: Stamp trial history series
if: always()
run: |
if [ -f /tmp/paid-report/trial-outcomes.jsonl ]; then
bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl
fi
# One merged receipt store per run: the plan's shipped set, every slice's
# new pass receipts, and the report's panel and negative receipts. Saved
# last, so the next planner restores it as the newest prefix match.
- name: Merge this run's receipts
if: always() && github.event_name == 'pull_request'
run: |
bun --no-install run scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache \
/tmp/paid-report/receipts /tmp/paid-report/report-receipts /tmp/paid-report/paid-slice-*/receipts
- name: Save this PR's verified judge and E2E results
if: always() && github.event_name == 'pull_request'
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
with:
path: /tmp/gstack-eval-input-cache
key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged
- name: Upload reconciliation output for the comment job
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: report-verdict
name: report-verdict-a${{ github.run_attempt }}
path: |
/tmp/report.txt
/tmp/paid-report/collector-outcomes.json
/tmp/paid-report/report-summary.md
if-no-files-found: ignore
retention-days: 30
- name: Upload trial outcomes for pass-rate history
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: trial-outcomes-pr-a${{ github.run_attempt }}
path: /tmp/paid-report/trial-outcomes.jsonl
if-no-files-found: ignore
retention-days: 90
- name: Fail the workflow when reconciliation failed
if: steps.reconcile.outputs.exit != '0'
run: exit 1
@@ -426,18 +455,13 @@ jobs:
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
pattern: paid-slice-[0-9]*
path: /tmp/paid-report
merge-multiple: true
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
name: report-verdict
name: report-verdict-a${{ github.run_attempt }}
path: /tmp/verdict
continue-on-error: true
# Verified counts come from the read-only report job, not repo code in
# this write-token job. Keeps the
# Every count, verdict and failure line comes from the read-only report
# job's collector-outcomes v2 (panelVerdict() ran there); this job runs
# no repo code and never recomputes a verdict. Keeps the
# "## E2E Evals" marker so the upsert keeps updating the same comment.
# Runs even when reconciliation failed — a red lane on the PR is the point.
- name: Post PR comment
@@ -446,13 +470,14 @@ jobs:
RECONCILE_EXIT: ${{ needs.slices-report.outputs.reconcile-exit }}
run: |
# shellcheck disable=SC2086,SC2059
RESULTS=$(find /tmp/paid-report -name '*.json' ! -name 'manifest.json' ! -name 'slice-*.json' ! -name '_partial*' 2>/dev/null | sort)
TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; FLAKY=0; EXECUTED=0; REUSED=0; COST="0"
TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; EXECUTED=0; REUSED=0; COST="0"
SUITE_LINES=""
VERIFIED=/tmp/verdict/paid-report/collector-outcomes.json
if ! jq -e '
. as $summary |
.version == 1 and (.files | type == "array") and (.totals | type == "object") and
.version == 2 and (.files | type == "array") and (.totals | type == "object") and
(.headline | type == "array") and (.failures | type == "array") and (.panels | type == "array") and
(.verdict.verdict == "GREEN" or .verdict.verdict == "RED") and
([.files[] | .total == (.passed + .failed + .manual_accepted) and
(.total == (.executed + .reused)) and
([.total,.passed,.failed,.manual_accepted,.executed,.reused,.attempts,.flaky] | all(. >= 0 and (floor == .))) ] | all) and
@@ -461,100 +486,69 @@ jobs:
all(. as $key | ([$summary.files[] | .[$key]] | add // 0) == $summary.totals[$key]))
' "$VERIFIED" >/dev/null 2>&1; then
VERIFIED=""
echo 'Verified collector summary unavailable; manual acceptance is unavailable/unverified.'
echo 'Verified report summary unavailable; no verdict, counts or manual acceptance can be shown.'
fi
HEADLINE='(no verified report headline)'
FAILURES=""
if [ -n "$VERIFIED" ]; then
while IFS=$'\t' read -r f T P F M FL EX RE _ATTEMPTS C TIER SHARD; do
[ "$T" -eq 0 ] && continue
TOTAL=$((TOTAL + T)); PASSED=$((PASSED + P)); FAILED=$((FAILED + F))
MANUAL=$((MANUAL + M)); FLAKY=$((FLAKY + FL))
MANUAL=$((MANUAL + M))
EXECUTED=$((EXECUTED + EX)); REUSED=$((REUSED + RE))
COST=$(echo "$COST + $C" | bc)
STATUS_ICON="✅"
[ "$M" -gt 0 ] && STATUS_ICON="⚠ manual/unscored"
[ "$F" -gt 0 ] && STATUS_ICON="❌"
[ "$F" -eq 0 ] && [ "$M" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠"
SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | ${M} | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n"
done < <(jq -r '.files[] | [.file,.total,.passed,.failed,.manual_accepted,.flaky,.executed,.reused,.attempts,.cost,.tier,.shard] | @tsv' "$VERIFIED")
else
for f in $RESULTS; do
if ! jq -e '.total_tests' "$f" >/dev/null 2>&1; then
echo "Skipping malformed JSON: $f"
continue
fi
# FINAL-attempt accounting: eval-store keeps EVERY retry attempt
# as its own record (that's the flake telemetry), so counting raw
# records marks a pass-on-retry as a failure and inflates totals.
# Group by test name and judge the LAST record. Retry metadata
# includes both passing and failing final outcomes; show it separately.
# Guarded: a file with total_tests but a null/non-array `tests`
# passes the -e probe, the group_by then fails, and an empty $T
# would abort the whole step under bash -e ([ "" -eq 0 ] is an
# error) — killing the comment on exactly the corrupted-artifact
# runs where the red evidence matters (claude adversarial).
STATS=$(jq -r '[.tests | group_by(.name)[] | last] as $final | "\($final | length) \([$final[] | select(.passed)] | length) \([$final[] | select(.passed | not)] | length) \(.flaky_retries // [] | length) \([$final[] | select(.execution != "reused")] | length) \([$final[] | select(.execution == "reused")] | length)"' "$f" 2>/dev/null) || { echo "Skipping malformed tests[] in: $f"; continue; }
read -r T P F FL EX RE <<< "$STATS"
[ -z "$T" ] && { echo "Skipping malformed tests[] in: $f"; continue; }
C=$(jq -r '.total_cost_usd // 0' "$f")
TIER=$(jq -r '.tier // "unknown"' "$f")
SHARD=$(jq -r '.shard // "-"' "$f")
[ "$T" -eq 0 ] && continue
TOTAL=$((TOTAL + T))
PASSED=$((PASSED + P))
FAILED=$((FAILED + F))
FLAKY=$((FLAKY + FL))
EXECUTED=$((EXECUTED + EX))
REUSED=$((REUSED + RE))
COST=$(echo "$COST + $C" | bc)
STATUS_ICON="✅"
[ "$F" -gt 0 ] && STATUS_ICON="❌"
[ "$F" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠"
SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | unverified | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n"
done
# Report-sanitized lines (no @-mentions, one capped line each), fenced here.
HEADLINE=$(jq -r '.headline[]' "$VERIFIED")
FAILURES=$(jq -r '.failures[]' "$VERIFIED")
fi
COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.'
STATUS="✅ PASS"
if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ]; then STATUS="❌ FAIL"; fi
if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ] \
|| { [ -n "$VERIFIED" ] && [ "$(jq -r '.verdict.verdict' "$VERIFIED")" != "GREEN" ]; }; then STATUS="❌ FAIL"; fi
if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi
if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (manual acceptance unavailable/unverified)'; fi
if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (verified report unavailable)'; fi
BODY="## E2E Evals: ${STATUS}
**${PASSED} automated passed / ${TOTAL} final results** | **${FAILED} failed, ${MANUAL} manual accepted (unscored; no score-cache credit)** | **${EXECUTED} executed, ${REUSED} reused** | **\$${COST}** total cost | reconcile exit: ${RECONCILE_EXIT:-missing}$([ "$FLAKY" -gt 0 ] && printf ' | ⚠ %s cases with multiple attempts' "$FLAKY")
\`\`\`
${HEADLINE}
\`\`\`
**${EXECUTED} executed, ${REUSED} reused** rule/judge records | **${MANUAL} manual accepted (unscored; no score-cache credit)** | **\$${COST}** rule/judge cost | reconcile exit: ${RECONCILE_EXIT:-missing}
${COVERAGE}
<details><summary>Rule and judge shards</summary>
| Shard | Automated result | Manual/unscored | Executed | Reused | Status | Cost |
|-------|------------------|-----------------|----------|--------|--------|------|
$(echo -e "$SUITE_LINES")
</details>
<details><summary>Fail-closed reconciliation</summary>
\`\`\`
$(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null || echo '(no reconciliation output)')
$(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo '(no reconciliation output)')
\`\`\`
</details>
---
*Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → duration-packed executors → fail-closed report). Reused results retain their original provenance and expiry.*"
*Sliced lane: planner → duration-packed executors → fail-closed report. Behavior cases run a pre-registered 3-trial panel (PASS at 2/3 with no contract violation); a PASS 2/3 is shown with its failed trial, never as a clean pass. Reused results retain their original provenance and expiry.*"
if [ "$FAILED" -gt 0 ]; then
FAILURES=""
for f in $RESULTS; do
if ! jq -e '.failed' "$f" >/dev/null 2>&1; then continue; fi
if [ -n "$VERIFIED" ]; then
FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false and (has("manual_review") | not))][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error")
else
FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false)][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error")
fi
FAILURES="${FAILURES}${FAILS}\n"
done
if [ -n "$FAILURES" ]; then
BODY="${BODY}
### Failures
$(echo -e "$FAILURES")"
### Failures and split verdicts
\`\`\`
${FAILURES}
\`\`\`"
fi
COMMENT_ID=$(gh api repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/comments \
+13 -5
View File
@@ -148,7 +148,8 @@ When fixing failures or preparing `/ship`, follow this order:
public events in free regressions, including negative controls, before paying
for another agent run. Check behavior and acknowledgments; match exact prose
only when that prose is the contract. Do not lower thresholds, increase model
budgets, skip cases, or rejudge a failure to manufacture a pass.
budgets, skip cases, or rejudge a failure to manufacture a pass. A
pre-registered fixed panel is not rejudging.
For policy or validation repairs, exercise the actual registered callback with
representative native input and assert that it uses the helper’s result.
When renderer or parser failures recur at the same boundary, verify the
@@ -208,10 +209,16 @@ When fixing failures or preparing `/ship`, follow this order:
result and pending permission state; diagnose a blocked actor before waiting
through its deadline. Preserve cancellation separately from a test verdict.
Skipped or unstarted cases
do not satisfy coverage; preserve every attempt. Retries follow the approved
policy in `test/helpers/eval-budgets.ts`: a timed-out attempt is a verdict, so
only files whose every case budget is CAPTURE tier or shorter keep one retry;
never add retries to pass a longer case.
do not satisfy coverage; preserve every attempt. Paid evals never retry. Each
case's kind (`E2E_KINDS`) fixes its trials before the run: `rule` one trial;
`behavior` a panel of 3 independent trials, PASS at >= 2 with no contract
violation; `judge` 3 samples on one output, gated on the mean against the
unchanged threshold. Never add trials, samples or dispatches after seeing a
result, never change a kind to change a verdict without pass-rate evidence,
and report every trial. Quarantine follows `CASE_QUARANTINE`'s entry and exit
rules only (`EVAL_POLICY`, `docs/TESTING_INTERNALS.md`). A census whose every
red is machine-classified INFRA or INCOMPLETE may be re-dispatched once as a
new run; report both runs.
7. Prove all known repairs with focused tests, including affected paid cases.
Rerun a failed case only after a concrete repair or a demonstrated launch
correction. Run the remaining required selected evaluations on the integrated
@@ -245,6 +252,7 @@ bun run test # complete free suite via the strict shard runner (no A
bun run test:ubicloud # same suite on an ephemeral 16-vCPU Ubicloud VM (needs UBICLOUD_API_KEY)
bun run eval:bg:pr # changed fast live probes + selected judges, with explicit deferrals
bun run eval:bg:release # fresh complete gate + periodic live coverage
bun run eval:pass-rates # per-case trial pass rates (Wilson), drift and quarantine alarms (--case, --gate)
bun run scripts/test-paid-shards.ts --tier periodic --list --slice-budget 540 --jobs 2 # CI slice plan preview (free)
bun run test:windows # curated Windows-safe subset (runs on windows-latest)
bun run build # generate docs + compile binaries
+45 -6
View File
@@ -239,10 +239,29 @@ Complete start-to-finish flows belong to the `marathon` tier
(`describeE2ETier('marathon')`), which runs only in the non-blocking
`evals-marathon.yml` lane (weekly and on dispatch) and never gates a merge.
Retries: a timed-out attempt is a verdict. Only files whose every case budget is
CAPTURE tier or shorter (`RETRY_MAX_CASE_MS` in `test/helpers/eval-budgets.ts`)
keep one automatic retry for fast-failing flakes; every other paid file runs once.
Case budgets themselves never change with this rule.
Verdicts: paid evals never retry. Each case's kind in `E2E_KINDS`
(`test/helpers/touchfiles-data.ts`) fixes its trials before the run, from the
constants in `EVAL_POLICY` (`test/helpers/periodic-exclude-data.ts`):
- `rule` (the default): one trial; any failed assertion fails the case. Use it
when nothing stochastic decides the verdict, or when the verdict checks a
contract the product must meet every run (no writes in plan mode, a question
before a decision, a skill-mandated step, no leaked secret).
- `behavior`: a panel of 3 independent trials run as parallel case shards,
PASS at 2 or more with no contract violation (`expectContract()`). Use it only
when a live model choice decides the verdict and an occasional deviation is
acceptable product behavior; the one-line reason goes in `BEHAVIOR_WHY`.
- `judge`: an LLM judge scoring a fixed input; 3 samples of the same prompt,
gated on the per-dimension mean (booleans on a majority) against the
unchanged threshold. An erroring sample fails the panel and is never resampled.
A timed-out, crashed or infrastructure-failed trial counts as a failed trial and
is reported with its class; a missing trial makes the case INCOMPLETE, which
fails the lane. A 2-of-3 pass is reported as `PASS 2/3` with the failed trial's
cause, never as a clean pass. Case budgets and thresholds never change with
this policy. Quarantine (`CASE_QUARANTINE`) and history are described in
`docs/TESTING_INTERNALS.md`; `bun run eval:pass-rates --case <id>` shows a
case's per-trial pass rate with its Wilson interval.
CI enables verified first-attempt reuse for 16 workflow quality judges for
24 hours within the same PR. The cookie workflow's custom input and the other 11
@@ -415,7 +434,7 @@ When E2E tests run, they produce machine-readable artifacts in `~/.gstack-dev/`:
bun run eval:list # list all eval runs (turns, duration, cost per run)
bun run eval:compare # compare two runs — shows per-test deltas + Takeaway commentary
bun run eval:summary # aggregate stats + per-test efficiency averages across runs
bun run eval:flake-rank # rank tests by flake signal: retried passes first, then failure rate (--json, --dir, --since-days)
bun run eval:pass-rates # per-case trial pass rates + Wilson intervals from recent weekly runs (--case, --runs, --dir, --backfill, --json, --gate); eval:flake-rank is an alias
```
**Detached runs for agents and long suites.** When an agent (or you, for a run
@@ -463,7 +482,9 @@ Override the judge model per run with `GSTACK_EVAL_MODEL_JUDGE`:
- **Completeness** — Are all commands, flags, and usage patterns documented?
- **Actionability** — Can the agent execute tasks using only the information in the doc?
Each dimension is scored 1-5. Threshold: every dimension must score **≥ 4**. There's also a regression test that compares generated docs against the hand-maintained baseline from `origin/main` — generated must score equal or higher.
Each dimension is scored 1-5 by a panel of 3 samples of the same prompt, drawn
concurrently; each dimension's panel mean must meet that judge's threshold (≥ 4
for most dimensions; see each case). An erroring sample fails the panel. There's also a regression test that compares generated docs against the hand-maintained baseline from `origin/main` — generated must score equal or higher.
```bash
# Needs ANTHROPIC_API_KEY in .env — included in bun run test:evals
@@ -483,6 +504,24 @@ fails, add the named path to the named key and check selection with
`bun run scripts/test-paid-shards.ts --tier gate --profile pr --list`. The rule is a lower bound: a fixture
path the test builds at runtime is not visible to it, so add such paths to the key by hand.
### Add a paid eval
1. **Test file.** Write the case in a paid test file, registered with a literal
name (`testIfSelected('<case-id>', ...)`), grading the outcome (files, git
state, native questions, exit status) rather than wording, unless the step
itself is the contract. Wrap contract assertions in `expectContract()`.
2. **Touchfiles.** Add `'<case-id>': [...]` to `E2E_TOUCHFILES`; `bun test
test/touchfiles.test.ts` names any missing closure path.
3. **Tier.** Add it to `E2E_TIERS`: `gate` for cheap contracts every PR needs,
`periodic` for long or model-quality cases, `marathon` for complete flows.
4. **Kind.** Add it to `E2E_KINDS` (`rule` unless a live model choice may
acceptably deviate; then `behavior` plus a `BEHAVIOR_WHY` line).
`bun test test/eval-kinds.test.ts` prints the literal to add.
5. **PR profile.** If a PR should run it, add it to `scripts/test-pr-profile.ts`
and check `bun run scripts/test-paid-shards.ts --tier gate --profile pr --list`.
6. **Try the panel locally.** `bun run scripts/test-paid-shards.ts --tier <tier>
--case <case-id> --trials 3` runs the same panel CI runs, before you push.
### CI
A GitHub Action (`.github/workflows/skill-docs.yml`) generates all hosts on pushes to main and on PRs, then rejects tracked differences and nonignored untracked output. Generation errors also fail the job. Optional ignored host caches are not compared against Git.
+2 -2
View File
@@ -1214,7 +1214,7 @@ Binary Images:
{ status: 1, stdout: '', stderr: '' }, { status: 0, stdout: '', stderr: '' },
{ status: 0, stdout: 'truncated-private-row', stderr: '' },
{ status: null, stdout: null, stderr: null, error: new Error('synthetic-private-error') },
]) expect(inspectUidProcesses(23456, performance.now() + 10_000, {}, (() => result) as typeof spawnSync)).toEqual({ available: false });
]) expect(inspectUidProcesses(23456, performance.now() + 10_000, {}, (() => result) as unknown as typeof spawnSync)).toEqual({ available: false });
});
test('numeric UID process filtering runs through the real global process table', () => {
@@ -1251,7 +1251,7 @@ Binary Images:
{ status: 113, stdout: '', stderr: 'Could not find domain for user uid: 23456' },
{ status: null, stdout: null, stderr: null, error: new Error('synthetic-private-error') },
]) {
const observation = inspectUserDomain(23456, performance.now() + 10_000, {}, (() => result) as typeof spawnSync);
const observation = inspectUserDomain(23456, performance.now() + 10_000, {}, (() => result) as unknown as typeof spawnSync);
expect(observation.state).toBe('unavailable');
expect(observation.structure).toBeUndefined();
expect(JSON.stringify(observation)).not.toContain('synthetic-private');
+124 -16
View File
@@ -252,24 +252,20 @@ processes × `EVALS_CONCURRENCY` within-shard, per-shard `GSTACK_EVAL_DIR`,
full-stream spooling to per-shard log files (path printed at START and on
failure), never-started/timed-out taxonomy, and parent-computed diff
selection propagated to children via `EVALS_SELECTION_JSON` (fail-open: a
child that can't parse it recomputes locally with one warning). Retries follow
one rule (`retriesForFiles`, `RETRY_MAX_CASE_MS` in `test/helpers/eval-budgets.ts`):
a timed-out attempt is a verdict, so a file keeps one Bun retry only when every
case budget is CAPTURE tier plus its recording grace or shorter (registered rows
derive it from `caseMs`, `SHORT_CASE_RETRY_FILES` lists the rest); every other
file, including the former `retries: 2` matrix rows, runs once. `--list` prints
each shard's retries. Files in `CASE_SHARDED_FILES` run one registered case per
child that can't parse it recomputes locally with one warning). Paid evals
never retry; each case's kind fixes its trials before the run (see "Eval verdict
policy" below). Files in `CASE_SHARDED_FILES` run one registered case per
process (`<file>#<case id>`, an exact `--test-name-pattern`, exactly one executed
case), so a long file of short cases spreads across runners and each case gets
its own SDK semaphore.
Flake telemetry rides the store: every recorded test carries its 1-based
`attempt` (a pass-on-attempt-2 stays visible forever — bun's own stream hides
it), runs list `flaky_retries`, the report warns on passed-only-on-retry
tests, and `bun run eval:flake-rank` ranks the series (retried passes first,
then failure rate; 60-day recency bound on eval files; the free lane's flake
ledger is folded in from `flakeLedgerPath()` — override with
`GSTACK_FLAKE_LEDGER`, the same env var the CI free lane sets before
uploading the ledger as the `flake-ledger` artifact). Census integrity is
Trial telemetry rides the store: every recorded test carries its 1-based
`attempt` plus, on an isolated trial shard, its `case_id`, `kind`, `trial`,
`panel` and `policy_version`, and each lane's report uploads one
`trial-outcomes` JSONL line per trial. `bun run eval:pass-rates`
(`eval:flake-rank` is an alias) turns that history into per-case pass rates
(see "Pass-rate history" below; the free lane's flake ledger is folded in from
`flakeLedgerPath()` — override with `GSTACK_FLAKE_LEDGER`, the same env var the
CI free lane sets before uploading the ledger as the `flake-ledger` artifact). Census integrity is
enforced from the free suite: every `E2E_TOUCHFILES` / `LLM_JUDGE_TOUCHFILES`
key must name a living paid test (`test/touchfiles.test.ts`'s reverse
invariant), and `git show <sha>:path` fixtures are banned — vendor the bytes
@@ -313,7 +309,7 @@ CI supplies the scoped cache/runtime configuration; local runs are fresh by
default. Cached scores must
pass current assertions; reused records retain their original source and time
and cannot renew the receipt. `scripts/e2e-shard-reuse.ts` extends the same receipts
to PR-profile E2E shards that run with zero retries (so the pass is structurally a
to PR-profile E2E shards (paid evals never retry, so a pass is structurally a
first attempt): the identity hashes the test's import closure, every tracked file
matched by the touchfiles of every case the file registers plus the global
touchfiles, the runner/workflow/setup actions, the child's environment pins, the
@@ -377,6 +373,118 @@ the runner parent and handed to shard children as `GSTACK_CLAUDE_CLI_VERSION`
(never spawned on a test thread), so a TUI-drift flake hunt is a grep, not
archaeology.
**Eval verdict policy** (`EVAL_POLICY` version 1 in
`test/helpers/periodic-exclude-data.ts`, pre-registered 2026-09-29). Paid evals
never retry. Each live case has exactly one kind in `E2E_KINDS`
(`test/helpers/touchfiles-data.ts`; `test/eval-kinds.test.ts` enforces coverage),
and the kind fixes its trials before the run:
- `rule` (default): one trial; any failed assertion fails the verdict. For
cases where nothing stochastic decides the verdict, or where it checks a
contract the product must meet every run.
- `behavior`: a panel of `n = 3` independent trials, launched together as
isolated case shards on different slices (key `<file>#<id>~t<N>`). All three
always run: no early stop and no conditional extra trial. PASS when at least
`k = 2` pass and no trial violated a contract (`expectContract()` stamps
`failure_class: 'contract'`). Each behavior case names its tolerated deviation
in `BEHAVIOR_WHY` and must have a literal registration so it can run alone.
- `judge`: an LLM judge scoring a fixed input. `judgePanel()`
(`test/helpers/llm-judge.ts`) draws 3 samples of the same prompt concurrently
inside the unchanged `JUDGE_MS`; numeric dimensions gate on the per-dimension
mean against the unchanged threshold (no dimension compensates for another),
booleans on a strict majority. A sample that errors (refusal, truncation,
non-JSON, a malformed field) fails the panel and is never resampled; a
refusal counts as an unscored panel only when every sample refused.
`callJudge`'s 429 backoff happens before any model output and is transport,
not a verdict retry. The workflow-judge cache stores whole panels only.
`panelVerdict()` (`test/helpers/eval-store.ts`) is the single verdict
function the report, `collector-outcomes.json`, the PR comment and pass-rates
all use. A timed-out, crashed or infrastructure-failed trial is a failed trial
recorded with its class; a missing or duplicate trial record makes the verdict
INCOMPLETE, which fails the lane; a 2/3 PASS is shown as `PASS 2/3` with the
failed trial's cause. A manual re-run adds trials under a new run attempt and
never replaces the first attempt's verdict. A red census is never rerun on
unchanged inputs: each red is diagnosed as product, test/detector, harness or
infra and resolved by a concrete repair and a fresh census, or listed as a named
red. The one exception: a census whose every red verdict is machine-classified
INFRA or INCOMPLETE (missing slice artifact, runner loss, API error before the
first model turn) may be re-dispatched once as a new run, and both runs are
reported. Changing any `EVAL_POLICY` constant after seeing census results needs
Garry's re-approval, a `version` bump and a fresh census;
`test/periodic-exclude-policy.test.ts` pins the approved values.
**Quarantine** (`CASE_QUARANTINE`, same file). An entry needs: a per-trial rate
below 95% over at least 10 post-policy trials of the case's current input
identity (pre-policy backfill may justify only an initial entry, labeled as
such); a written diagnosis in `reason` whose `failureClass` is `detector`,
`harness` or `model-latency` (a product defect is fixed or listed as a named
red, never quarantined); unchanged case touchfiles in the change that adds it;
and an owner, tracking pointer, `enteredAt` date and measurable `exit`. A
quarantined case still runs its full panel and reports in every lane but cannot
fail it, except on a hard break (0 of n) or a contract violation, and it never
counts as passing coverage. At most 10% of a blocking tier (gate, periodic) may
be quarantined. The weekly report fails when an entry passes its exit rule (at
least 97% over at least 10 trials) without being removed, when an entry is 8
weekly runs old, or when a tier is over its cap.
**Pass-rate history** (`bun run eval:pass-rates`, `scripts/eval-flake-rank.ts`).
It reads the `trial-outcomes` artifact of the last N completed
`evals-periodic.yml` runs on the current branch and `main` (flags: `--case`,
`--runs N`, `--branch`, `--dir`, `--backfill`, `--json`, `--gate`) and prints
per-case per-trial pass rates with 95% Wilson intervals. A series is one case
under one input identity, the hash of its own touchfiles minus
`GLOBAL_TOUCHFILES` (harness edits do not restart it), per model, Claude CLI
version and policy version; a change starts a new series and older ones stay
visible. Labels: INCONCLUSIVE below 10 trials, BROKEN when the latest run is
0/n after a prior interval at or above 95%, FLAKY when failures leave the
interval straddling 95%, FAILING when the whole interval is below it, PASSING
otherwise. `--backfill` imports legacy slice artifacts as pre-policy trials
(first attempt only; a record that names no registry id is listed as
unattributed, never guessed); they are display-only. `--gate` (the weekly
report) fails with ACTION REQUIRED, on post-policy trials of the current series
only, when a non-quarantined blocking case meets the entry rule (proposing an
entry), when a `rule` case does (rule case behaving like behavior: fix or
reclassify), when a blocking case's current identity is significantly below its
previous one (one-sided Fisher exact, α = 0.05, at least 6 trials each side,
Holm-controlled across the cases tested), and on the quarantine rules above.
History that cannot be fetched fails the gate closed.
**The arithmetic.** With per-trial pass rate p, the chance a single case goes
red (a false red while the product works, the catch rate once it has
regressed):
| p | 1 trial | 2-of-3 panel |
|---|---|---|
| 0.99 | 1.0% | 0.03% |
| 0.95 | 5.0% | 0.72% |
| 0.90 | 10.0% | 2.8% |
| 0.70 | 30.0% | 21.6% |
| 0.30 | 70.0% | 78.4% |
The panel removes most false reds at healthy rates, but it catches a 0.95 → 0.70
regression in one run only 21.6% of the time (a single trial 30%, retry-until-green
3%), so drift detection is the history rule's job, not the per-run verdict's.
The Fisher alarm is weak at the minimum sample (5.4% power for 0.95 → 0.70 at
6 trials a side), and ten straight passes still leave a 72% Wilson lower bound:
after this policy lands, every series starts INCONCLUSIVE.
A lane is all green with probability Π p_rule × Π P(≥2 of 3 | p_behavior) ×
Π p_judge. For the current registry (PR gate worst case: 107 rule cases and 24
judges; weekly census: 190 rule, 22 behavior and 25 judge verdicts), with rule
and judge verdicts at p_rule:
| p_rule | full PR gate | weekly, behavior p = 0.90 | 0.95 | 0.97 |
|---|---|---|---|---|
| 0.99 | 26.8% | 6.2% | 9.8% | 10.9% |
| 0.995 | 51.9% | 18.2% | 29.0% | 32.1% |
| 0.999 | 87.7% | 43.2% | 68.7% | 76.1% |
The rule term dominates: a green lane on a working product needs rule cases to
be near-deterministic (0.999), which is why failing detectors are converted to
outcome checks and product defects are fixed or named, and why each census
reports its expected lane false-red from the measured rates.
**Timeout policy.** Paid tests use the tiers in
`test/helpers/eval-budgets.ts` (JUDGE/CAPTURE/CAPTURE_LONG/PTY/PTY_LONG);
`test/eval-budgets-policy.test.ts` pins that every tier fits the shard wall
+7 -6
View File
@@ -31,12 +31,12 @@
"test:free": "bun run scripts/test-free-shards.ts",
"test:windows": "bun run scripts/test-free-shards.ts --windows-only",
"test:ubicloud": "bash scripts/ubicloud/test-free.sh",
"test:evals": "EVALS=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts",
"test:evals:all": "EVALS=1 EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts",
"test:e2e": "EVALS=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts",
"test:e2e:all": "EVALS=1 EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts",
"test:gate": "EVALS=1 EVALS_TIER=gate bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts",
"test:periodic": "EVALS=1 EVALS_TIER=periodic EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts",
"test:evals": "EVALS=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts",
"test:evals:all": "EVALS=1 EVALS_ALL=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts",
"test:e2e": "EVALS=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts",
"test:e2e:all": "EVALS=1 EVALS_ALL=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts",
"test:gate": "EVALS=1 EVALS_TIER=gate bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts",
"test:periodic": "EVALS=1 EVALS_TIER=periodic EVALS_ALL=1 bun test --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts",
"test:gate:sharded": "bun run scripts/test-paid-shards.ts --tier gate",
"test:periodic:sharded": "EVALS_ALL=1 bun run scripts/test-paid-shards.ts --tier periodic",
"test:codex": "EVALS=1 bun test test/codex-e2e.test.ts test/codex-e2e-sol-scope.test.ts",
@@ -52,6 +52,7 @@
"eval:compare": "bun run scripts/eval-compare.ts",
"eval:summary": "bun run scripts/eval-summary.ts",
"eval:flake-rank": "bun run scripts/eval-flake-rank.ts",
"eval:pass-rates": "bun run scripts/eval-flake-rank.ts",
"eval:watch": "bun run scripts/eval-watch.ts",
"eval:select": "bun run scripts/eval-select.ts",
"analytics": "bun run scripts/analytics.ts",
+147 -2
View File
@@ -30,6 +30,8 @@ import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure,
type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from './eval-input-cache';
import { matchGlob } from '../test/helpers/test-selection';
import { E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from '../test/helpers/touchfiles-data';
import { EVAL_CACHE_MAX_AGE_MS as RECEIPT_MAX_AGE_MS } from './eval-input-cache';
import { panelVerdict, TRIAL_ENV, type EvalCaseKind, type PanelShape, type PanelTrial } from '../test/helpers/eval-store';
export interface E2EShardReuseRequest {
root: string;
@@ -50,6 +52,8 @@ export interface E2EShardReuseRequest {
profile: string;
/** The exact environment the child receives. */
env: NodeJS.ProcessEnv;
/** Isolated trial shard: its panel policy is part of the identity; the trial index is run-scoped. */
panel?: { kind: EvalCaseKind; panel: PanelShape; quarantined: boolean };
}
export interface E2EShardReuseHit { key: string; source: EvalPassingProof['source'] }
@@ -61,7 +65,7 @@ const HARNESS_FILES = ['scripts/test-paid-shards.ts', 'scripts/e2e-shard-reuse.t
const ENV_PREFIXES = ['EVALS_', 'GSTACK_', 'CLAUDE_', 'ANTHROPIC_', 'OPENAI_', 'GEMINI_', 'BUN_', 'NODE_', 'PLAYWRIGHT_'];
/** Run-scoped values: provenance or transport, never behavior. Selection is bound as case ids. */
const RUN_SCOPED_ENV = new Set(['EVALS_RUN_ID', 'GSTACK_EVAL_DIR', 'EVALS_CACHE_DIR', 'EVALS_CACHE_PR', 'EVALS_CACHE_REPOSITORY',
'EVALS_CACHE_RUNTIME_ID', 'EVALS_CACHE_PURPOSE', 'EVALS_SELECTION_JSON', 'EVALS_JUDGE_SELECTION_JSON']);
'EVALS_CACHE_RUNTIME_ID', 'EVALS_CACHE_PURPOSE', 'EVALS_SELECTION_JSON', 'EVALS_JUDGE_SELECTION_JSON', TRIAL_ENV.trial]);
const SECRET_ENV = /KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL/;
/** The reuse-relevant environment the child sees; secrets contribute presence only. */
@@ -126,7 +130,8 @@ export function e2eShardIdentity(request: E2EShardReuseRequest): { status: 'elig
coverage: { dependencies: 'complete', prompts: 'complete', environment: 'complete' }, unknownDependencies: [],
files,
prompts: Object.fromEntries(request.caseIds.map(id => [id, source])),
parameters: { rootPackage, key: request.key, caseIds: [...request.caseIds].sort(), casePattern: request.casePattern,
parameters: { rootPackage, key: request.panel ? request.key.replace(/~t\d+$/, '') : request.key,
...(request.panel ? { panel: { kind: request.panel.kind, n: request.panel.panel.n, k: request.panel.panel.k, quarantined: request.panel.quarantined } } : {}), caseIds: [...request.caseIds].sort(), casePattern: request.casePattern,
expectedCases: request.expectedCases, retries: request.retries, timeoutMs: request.timeoutMs,
withinShardConcurrency: request.withinShardConcurrency, tier: request.tier, profile: request.profile,
environment: e2eReuseEnvironment(env) },
@@ -156,7 +161,13 @@ const validResult = (identity: EvalInputIdentity, key: string) => (value: EvalCa
* whose inputs did not change during execution.
*/
export function prepareE2EShardReuse(request: E2EShardReuseRequest): {
/** The input identity key: recorded on the outcome so the report can store verdicts against it. */
inputKey: string;
lookup(): E2EShardReuseHit | null;
/** Trial shards only: this trial's record from a whole PASS panel receipt of the plan's receipts. */
lookupPanelTrial(trial: number): { hit: E2EShardReuseHit; trial: PanelTrial } | null;
/** True when the inputs are unchanged since `before` (the outcome may carry inputKey). */
unchanged(): boolean;
publish(): void;
} | null {
if (e2eReuseLaneProblem(request.env, 'pr') !== null) return null;
@@ -164,6 +175,19 @@ export function prepareE2EShardReuse(request: E2EShardReuseRequest): {
if (before.status !== 'eligible') return null;
const common = { cacheDir: request.env.EVALS_CACHE_DIR!, purpose: 'gate' as const };
return {
inputKey: before.identity.key,
unchanged() {
const after = e2eShardIdentity(request);
return after.status === 'eligible' && after.identity.key === before.identity.key;
},
lookupPanelTrial(trial) {
if (!request.panel) return null;
const receipt = readPanelReceipt(common.cacheDir, before.identity.key);
if (!receipt || receipt.case !== request.caseIds[0] || receipt.kind !== request.panel.kind
|| receipt.panel.n !== request.panel.panel.n || receipt.panel.k !== request.panel.panel.k) return null;
const record = receipt.trials.find(t => t.trial === trial);
return record ? { hit: { key: receipt.key, source: receipt.source }, trial: record } : null;
},
lookup() {
const found = lookupEvalInputCache({ ...common, identity: before.identity, validateResult: validResult(before.identity, request.key) });
return found.status === 'reused' ? { key: found.key, source: found.source } : null;
@@ -184,3 +208,124 @@ export function prepareE2EShardReuse(request: E2EShardReuseRequest): {
},
};
}
// ─── Panel receipts, negative receipts and the planner's receipt selection ──
//
// Reuse is decided by the planner, once per panel: it ships the plan a
// receipt set in which every panel receipt is a whole PASS panel from one run
// and no pass receipt has a newer FAIL for the same identity. Executors look
// up only that set, so every trial of a panel sees the same receipts. The
// report writes panel receipts (all n trials fresh, one identity) and
// negative receipts (FAIL verdicts) after the verdict is known.
export interface PanelReceipt {
schema: 1;
key: string;
case: string;
kind: EvalCaseKind;
panel: PanelShape;
trials: PanelTrial[];
source: { runId: string; revision: string; completedAt: number };
}
export interface NegativeReceipt { schema: 1; key: string; source: { runId: string; revision: string; completedAt: number } }
const RECEIPT_KEY = /^[a-f0-9]{64}$/;
const validSource = (source: any) => !!source && typeof source.runId === 'string' && /^[\w./-]{1,160}$/.test(source.runId)
&& typeof source.revision === 'string' && /^[a-f0-9]{40}$/.test(source.revision) && Number.isSafeInteger(source.completedAt) && source.completedAt > 0;
function readJson(file: string, maxBytes = 64 * 1024): any {
try {
const stat = fs.lstatSync(file);
if (!stat.isFile() || stat.size > maxBytes) return null;
return JSON.parse(fs.readFileSync(file, 'utf8'));
} catch { return null; }
}
/** A whole, unexpired PASS panel receipt for `key`, re-verified with panelVerdict(); else null. */
export function readPanelReceipt(cacheDir: string, key: string, now = Date.now()): PanelReceipt | null {
if (!RECEIPT_KEY.test(key)) return null;
const receipt = readJson(path.join(cacheDir, `${key}.panel.json`));
if (!receipt || receipt.schema !== 1 || receipt.key !== key || typeof receipt.case !== 'string' || !validSource(receipt.source)
|| receipt.source.completedAt > now || now - receipt.source.completedAt >= RECEIPT_MAX_AGE_MS || !Array.isArray(receipt.trials)) return null;
try {
const verdict = panelVerdict({ case: receipt.case, kind: receipt.kind, panel: receipt.panel,
trials: receipt.trials.map((t: PanelTrial) => ({ ...t, attempt: 1 })) });
if (verdict.status !== 'PASS' || verdict.trials.length !== receipt.panel.n) return null;
} catch { return null; }
const negative = readJson(path.join(cacheDir, `${key}.fail.json`));
if (negative && validSource(negative.source) && negative.source.completedAt >= receipt.source.completedAt) return null;
return receipt as PanelReceipt;
}
export function writePanelReceipt(dir: string, receipt: PanelReceipt): void {
fs.mkdirSync(dir, { recursive: true });
fs.writeFileSync(path.join(dir, `${receipt.key}.panel.json`), `${JSON.stringify(receipt)}\n`, { mode: 0o600 });
}
export function writeNegativeReceipt(dir: string, receipt: NegativeReceipt): void {
fs.mkdirSync(dir, { recursive: true });
fs.writeFileSync(path.join(dir, `${receipt.key}.fail.json`), `${JSON.stringify(receipt)}\n`, { mode: 0o600 });
}
const receiptTime = (file: string): number => {
const parsed = readJson(file);
return Number(parsed?.source?.completedAt ?? parsed?.proof?.source?.completedAt) || 0;
};
/**
* Planner-side selection: copy `from` into `to`, dropping every pass or panel
* receipt that has a same-or-newer negative receipt for its identity, and
* every panel receipt that is not a whole PASS panel. Workflow-judge and
* other receipts pass through for their own validation at lookup.
*/
export function selectPlanReceipts(from: string, to: string, now = Date.now()): { shipped: number; blocked: string[] } {
fs.mkdirSync(to, { recursive: true });
const blocked: string[] = [];
let shipped = 0;
let names: string[] = [];
try { names = fs.readdirSync(from).filter(name => name.endsWith('.json')); } catch { return { shipped, blocked }; }
for (const name of names) {
const file = path.join(from, name);
const [key, suffix] = [name.slice(0, 64), name.slice(64)];
const negative = RECEIPT_KEY.test(key) ? readJson(path.join(from, `${key}.fail.json`)) : null;
const newerFail = negative && validSource(negative.source) && negative.source.completedAt >= receiptTime(file);
if (suffix === '.panel.json' && (newerFail || !readPanelReceipt(from, key, now))) { blocked.push(name); continue; }
if (suffix === '.json' && newerFail) { blocked.push(name); continue; }
fs.copyFileSync(file, path.join(to, name));
shipped++;
}
return { shipped, blocked };
}
/** Merge receipt directories into one store, keeping the newest file per name. */
export function mergeReceiptDirs(out: string, dirs: string[]): number {
fs.mkdirSync(out, { recursive: true });
let merged = 0;
for (const dir of dirs) {
let names: string[] = [];
try { names = fs.readdirSync(dir).filter(name => name.endsWith('.json')); } catch { continue; }
for (const name of names) {
const source = path.join(dir, name);
const target = path.join(out, name);
if (!fs.lstatSync(source).isFile()) continue;
if (fs.existsSync(target) && receiptTime(target) >= receiptTime(source)) continue;
fs.copyFileSync(source, target);
merged++;
}
}
return merged;
}
if (import.meta.main) {
const [command, first, ...rest] = process.argv.slice(2);
if (command === 'select' && first && rest[0]) {
const result = selectPlanReceipts(first, rest[0]);
console.log(`[e2e-reuse] shipped ${result.shipped} receipt(s) to the plan; blocked ${result.blocked.length} (newer FAIL or partial panel)`);
} else if (command === 'merge' && first) {
console.log(`[e2e-reuse] merged ${mergeReceiptDirs(first, rest)} receipt(s) into ${first}`);
} else {
console.error('usage: bun run scripts/e2e-shard-reuse.ts select <from> <to> | merge <out> <dir...>');
process.exit(2);
}
}
+589 -34
View File
@@ -1,29 +1,55 @@
#!/usr/bin/env bun
/**
* eval-flake-rank — the flake-telemetry dial (WS1).
* eval-pass-rates (alias: eval-flake-rank) — per-case trial pass rates.
*
* Aggregates per-test series across every FINALIZED eval-store run on this
* machine (default: ~/.gstack/projects/<slug>/evals/, shard dirs included)
* plus the free suite's flake ledger, and ranks tests by flake signal:
* retried passes first (a test that needs attempt 2 to go green is the
* definition of a flake), then failure rate.
* Reads trial records (one JSONL line per trial: case, kind, trial, outcome,
* exit_reason, duration, cost, model, CLI version, series identity, run id,
* sha, policy_version) from the last N completed `evals-periodic.yml` runs on
* the current branch and `main` (downloading only each run's small
* `trial-outcomes` artifact through `gh`), plus any local eval dirs, and
* prints per-case per-trial pass rates with 95% Wilson intervals.
*
* This is the readable dial behind two policies:
* - a flaky pass never blocks a merge, but it is recorded and RANKED here;
* - the required-check promotion (WS16) needs weeks of clean flake-rank,
* not vibes.
* A series is one case under one input identity: the case's own touchfiles
* minus GLOBAL_TOUCHFILES (`caseSeriesIdentities`), grouped by model and CLI
* version, per policy_version. A new identity starts a new series; earlier
* series stay visible. Only post-policy trials of the current series feed the
* labels and alarms. Legacy eval-store records (`--backfill`, `--dir`) are
* imported as pre-policy trials (first attempt only; a missing attempt means
* 1) and are display-only.
*
* Labels: INCONCLUSIVE (below the entry rule's minimum trials), BROKEN (latest run 0/n
* after a prior interval at or above the entry rate), FLAKY (failures and an
* interval straddling the entry rate), FAILING (interval below the entry
* rate), PASSING (otherwise).
*
* The weekly gate (`--gate`) exits non-zero with ACTION REQUIRED when a
* non-quarantined case meets the quarantine entry rule, a rule case behaves
* like a behavior case, a blocking case's current-identity rate is
* significantly below its previous identity (one-sided Fisher exact,
* Holm-controlled across cases), or a CASE_QUARANTINE entry has met its exit
* rule, expired, or pushed its tier over the cap. History that cannot be
* fetched fails the gate closed.
*
* Usage:
* bun run eval:flake-rank # project eval dir
* bun run eval:flake-rank --dir <path> # e.g. downloaded CI artifacts
* bun run eval:flake-rank --json # machine-readable
* bun run eval:pass-rates # last 10 weekly runs, this branch + main
* bun run eval:pass-rates --case <id> --runs 20
* bun run eval:pass-rates --dir <path> # local eval dirs / downloaded artifacts (repeatable)
* bun run eval:pass-rates --backfill # also import legacy slice artifacts, labeled pre-policy
* bun run eval:pass-rates --json | --gate
*/
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { getProjectEvalDir, isPartialEval, isFinalizedEvalResultFile, type EvalResult } from '../test/helpers/eval-store';
import { evalEntryOutcome } from '../test/helpers/eval-store';
import { spawnSync } from 'node:child_process';
import { createHash } from 'node:crypto';
import { isPartialEval, isFinalizedEvalResultFile, evalEntryOutcome, failureClassOf, parseTrialOutcomes, sanitizeTrialError,
TRIAL_OUTCOME_SCHEMA, type EvalCaseKind, type EvalResult, type TrialOutcomeRecord } from '../test/helpers/eval-store';
import { flakeLedgerPath, type FlakeLedgerEntry } from './test-free-shards';
import { E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, GLOBAL_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from '../test/helpers/touchfiles-data';
import { CASE_QUARANTINE, EVAL_POLICY } from '../test/helpers/periodic-exclude-data';
import { matchGlob } from '../test/helpers/test-selection';
import { CASE_TEST_NAMES } from './test-paid-shards';
interface TestSeries {
name: string;
@@ -111,33 +137,561 @@ function readFreeLedger(): FlakeLedgerEntry[] {
return out;
}
// --- Trial records ---
/**
* A trial record as pass-rates reads it: eval-store's trial-outcomes schema
* plus the series identity the report job stamps (caseSeriesIdentities).
* policy_version 0 marks a pre-policy (backfilled) record.
*/
export type TrialRecord = TrialOutcomeRecord & { series_identity?: string };
/** Per-file cap for downloaded artifacts: pass-rates parses data only, never executes it. */
export const TRIAL_OUTCOMES_MAX_BYTES = 8 * 1024 * 1024;
/** Every `trial-outcomes*.jsonl` file under a directory, size-capped, schema-validated by eval-store. */
export function readTrialOutcomeDir(dir: string): { records: TrialRecord[]; errors: string[] } {
const records: TrialRecord[] = [];
const errors: string[] = [];
if (!fs.existsSync(dir)) return { records, errors };
for (const name of fs.readdirSync(dir, { recursive: true }) as string[]) {
if (!/(^|\/)trial-outcomes[^/]*\.jsonl$/.test(name)) continue;
const full = path.join(dir, name);
const parsed = parseTrialOutcomes(fs.readFileSync(full, 'utf8'), { maxBytes: TRIAL_OUTCOMES_MAX_BYTES });
records.push(...parsed.records.map(record => ({
...record, series_identity: typeof (record as TrialRecord).series_identity === 'string'
? (record as TrialRecord).series_identity!.slice(0, 64) : undefined })));
errors.push(...parsed.errors.map(error => `${full}: ${error}`));
}
return { records, errors };
}
// --- Registry attribution and series identity ---
export interface Registry {
kinds: Record<string, EvalCaseKind>;
tiers: Record<string, string>;
touchfiles: Record<string, string[]>;
judgeTouchfiles: Record<string, string[]>;
globals: readonly string[];
testNames: Record<string, string>;
}
export const LIVE_REGISTRY: Registry = {
kinds: E2E_KINDS, tiers: E2E_TIERS, touchfiles: E2E_TOUCHFILES, judgeTouchfiles: LLM_JUDGE_TOUCHFILES,
globals: GLOBAL_TOUCHFILES, testNames: CASE_TEST_NAMES,
};
/** A case's tier: its E2E_TIERS value, or 'judge' for an LLM-judge entry. */
export function caseTier(id: string, registry: Registry = LIVE_REGISTRY): string {
return registry.tiers[id] ?? (id in registry.judgeTouchfiles ? 'judge' : 'unknown');
}
/**
* Attribute a legacy eval-store record to a registry id: the case-shard slug
* suffix (`<file>--<id>`), the recorded name or its exact slug (`/qa b6-static`
* is `qa-b6-static`), a CASE_TEST_NAMES label, or the only id its shard file
* registers. Anything else is unattributed (null).
*/
export function attributeLegacyRecord(name: string, shard: string | undefined, registry: Registry = LIVE_REGISTRY): string | null {
const known = (id: string) => id in registry.kinds;
const [slugFile, slugCase] = (shard ?? '').split('--');
if (slugCase && known(slugCase)) return slugCase;
if (known(name)) return name;
const slug = name.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '');
if (known(slug)) return slug;
const labeled = Object.entries(registry.testNames).find(([, label]) => label === name)?.[0];
if (labeled && known(labeled)) return labeled;
if (slugFile) {
const file = `test/${slugFile}.test.ts`;
const owners = Object.keys(registry.touchfiles).filter(id => registry.touchfiles[id]!.includes(file));
if (owners.length === 1 && known(owners[0]!)) return owners[0]!;
}
return null;
}
/**
* Series identity per case: a hash of the git blob ids of the files matching
* the case's own touchfiles, excluding GLOBAL_TOUCHFILES (harness edits are
* markers, not new series). The report job stamps this on every trial record.
*/
export function caseSeriesIdentities(ids: string[], root: string, registry: Registry = LIVE_REGISTRY): Record<string, string> {
const listed = spawnSync('git', ['ls-files', '-s'], { cwd: root, encoding: 'utf8', timeout: 20_000, maxBuffer: 64 * 1024 * 1024 });
if (listed.status !== 0) throw new Error(`git ls-files failed: ${listed.stderr}`);
const blobs = listed.stdout.split('\n').filter(Boolean).map(line => {
const [meta, file] = line.split('\t');
return { file: file!, blob: meta!.split(' ')[1]! };
}).filter(entry => !registry.globals.some(pattern => matchGlob(entry.file, pattern)));
return Object.fromEntries(ids.map(id => {
const patterns = registry.touchfiles[id] ?? registry.judgeTouchfiles[id] ?? [];
const lines = blobs.filter(entry => patterns.some(pattern => matchGlob(entry.file, pattern)))
.map(entry => `${entry.file} ${entry.blob}`).sort();
return [id, createHash('sha256').update(`${id}\n${lines.join('\n')}`).digest('hex').slice(0, 16)];
}));
}
/**
* Import legacy eval-store result files as pre-policy trials (policy_version
* 0, source 'backfill'): first attempt only (a missing attempt means 1),
* attributed by registry id, never guessed. A manual-review acceptance carries
* no automated verdict: it is counted and shown, never scored. Without a CI
* run, each local result file is its own run.
*/
export function backfillEvalFiles(files: string[], run?: { run_id: string; sha?: string; timestamp?: string },
registry: Registry = LIVE_REGISTRY): { records: TrialRecord[]; unattributed: string[]; manualReviews: string[] } {
const records: TrialRecord[] = [];
const unattributed = new Set<string>();
const manualReviews: string[] = [];
for (const file of files) {
let result: EvalResult & { shard?: string; claude_cli_version?: string };
try { result = JSON.parse(fs.readFileSync(file, 'utf8')); } catch { continue; }
if (isPartialEval(result, file) || !Array.isArray(result.tests)) continue;
const seen = new Set<string>();
for (const entry of result.tests) {
if ((entry.attempt ?? 1) !== 1 || seen.has(entry.name)) continue;
seen.add(entry.name);
const id = attributeLegacyRecord(entry.name, result.shard, registry);
if (!id) { unattributed.add(entry.name); continue; }
const outcome = evalEntryOutcome(entry);
if (outcome === 'manual-review') { manualReviews.push(id); continue; }
records.push({
schema: TRIAL_OUTCOME_SCHEMA, case: id,
file: result.shard ? `test/${result.shard.split('--')[0]}.test.ts` : 'unknown',
tier: caseTier(id, registry), kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1,
outcome, ...(outcome === 'failed' ? { failure_class: failureClassOf(entry) } : {}),
exit_reason: entry.exit_reason, error: sanitizeTrialError(entry.error),
duration_ms: Math.max(0, entry.duration_ms || 0), cost_usd: Math.max(0, entry.cost_usd || 0),
model: entry.model, cli_version: result.claude_cli_version, policy_version: 0, quarantined: false,
execution: entry.execution === 'reused' ? 'reused' : 'executed', source: 'backfill',
run_id: run?.run_id ?? `local:${file}`, sha: run?.sha ?? result.git_sha, recorded_at: run?.timestamp ?? result.timestamp,
});
}
}
return { records, unattributed: [...unattributed].sort(), manualReviews };
}
// --- Statistics ---
/** 95% Wilson score interval for k successes in n trials. */
export function wilsonInterval(k: number, n: number, z = 1.96): { lo: number; hi: number } {
if (n <= 0) return { lo: 0, hi: 1 };
const p = k / n, z2 = z * z, denom = 1 + z2 / n;
const center = (p + z2 / (2 * n)) / denom;
const half = (z * Math.sqrt(p * (1 - p) / n + z2 / (4 * n * n))) / denom;
return { lo: Math.max(0, center - half), hi: k === n ? 1 : Math.min(1, center + half) };
}
function logChoose(n: number, k: number): number {
let sum = 0;
for (let i = 1; i <= k; i++) sum += Math.log(n - k + i) - Math.log(i);
return sum;
}
/**
* One-sided Fisher exact p-value that the CURRENT pass rate is below the
* PREVIOUS one: P(X <= curPass) under the hypergeometric null with the
* observed margins.
*/
export function fisherOneSidedLower(curPass: number, curN: number, prevPass: number, prevN: number): number {
const passes = curPass + prevPass, total = curN + prevN;
const denom = logChoose(total, passes);
let p = 0;
for (let x = Math.max(0, passes - prevN); x <= curPass; x++) p += Math.exp(logChoose(curN, x) + logChoose(prevN, passes - x) - denom);
return Math.min(1, p);
}
/** Holm step-down: the indices whose p-values are rejected at family-wise alpha. */
export function holmRejections(pValues: number[], alpha: number): Set<number> {
const order = pValues.map((p, index) => ({ p, index })).sort((a, b) => a.p - b.p);
const rejected = new Set<number>();
for (let rank = 0; rank < order.length; rank++) {
if (order[rank]!.p > alpha / (order.length - rank)) break;
rejected.add(order[rank]!.index);
}
return rejected;
}
// --- Analysis ---
export type PassRateLabel = 'INCONCLUSIVE' | 'BROKEN' | 'FLAKY' | 'FAILING' | 'PASSING';
export type AlarmKind = 'drift' | 'rule-as-behavior' | 'regression' | 'quarantine-exit' | 'quarantine-expired'
| 'quarantine-cap' | 'quarantine-invalid';
/** The EVAL_POLICY fields pass-rates reads (structural, so tests can vary them). */
export interface PassRatePolicy {
version: number;
quarantine: { entry: { rate: number; minTrials: number }; exit: { rate: number; minTrials: number }; capFraction: number; expiryWeeklyRuns: number };
drift: { fisherAlpha: number; fisherMinPerSide: number };
}
export type QuarantineEntry = (typeof CASE_QUARANTINE)[string];
/** Tiers whose cases block a lane; quarantine applies only to them. */
export const BLOCKING_TIERS: readonly string[] = ['gate', 'periodic'];
const QUARANTINE_FAILURE_CLASSES: readonly string[] = ['detector', 'harness', 'model-latency'];
export interface SeriesStats {
key: string;
identity: string;
model: string;
cli: string;
policyVersion: number;
passes: number;
/** Scored trials: passed + failed (skipped trials carry no verdict). */
trials: number;
infra: number;
interval: { lo: number; hi: number };
firstSeen: string;
lastSeen: string;
runs: string[];
}
export interface CasePassRate {
case: string;
kind: EvalCaseKind;
tier: string;
quarantined: boolean;
label: PassRateLabel;
/** Manual-review acceptances: visible, never scored. */
manualReviews: number;
current: SeriesStats | null;
previous: SeriesStats | null;
prePolicy: SeriesStats | null;
series: SeriesStats[];
latestRun: { runId: string; passes: number; trials: number } | null;
}
export interface Alarm { kind: AlarmKind; case: string; message: string }
export interface PassRateReport {
policyVersion: number;
cases: CasePassRate[];
alarms: Alarm[];
postPolicyTrials: number;
prePolicyTrials: number;
unattributed: string[];
errors: string[];
}
export interface AnalyzeOptions {
registry?: Registry;
quarantine?: Record<string, QuarantineEntry>;
policy?: PassRatePolicy;
/** Completed weekly-run timestamps in the window, for quarantine expiry. */
weeklyRuns?: string[];
now?: number;
unattributed?: string[];
errors?: string[];
manualReviews?: string[];
}
const at = (record: TrialRecord) => record.recorded_at ?? '';
const runOf = (record: TrialRecord) => `${record.run_id ?? record.sha ?? 'local'}#${record.attempt}`;
function seriesStats(key: string, records: TrialRecord[]): SeriesStats {
const scored = records.filter(record => record.outcome !== 'skipped');
const passes = scored.filter(record => record.outcome === 'passed').length;
const times = records.map(at).sort();
const first = records[0]!;
return {
key, identity: first.series_identity ?? 'unknown', model: first.model ?? 'unknown', cli: first.cli_version ?? 'unknown',
policyVersion: first.policy_version, passes, trials: scored.length,
infra: scored.filter(record => record.outcome === 'failed' && record.failure_class === 'infra').length,
interval: wilsonInterval(passes, scored.length), firstSeen: times[0] ?? '', lastSeen: times[times.length - 1] ?? '',
runs: [...new Set(records.map(runOf))],
};
}
/** Weekly runs completed after an entry's enteredAt; offline, whole weeks elapsed. */
export function quarantineRunsSince(enteredAt: string, weeklyRuns: string[] | undefined, now: number): number {
const entered = Date.parse(enteredAt);
if (!Number.isFinite(entered)) return Number.POSITIVE_INFINITY;
if (weeklyRuns && weeklyRuns.length) return weeklyRuns.filter(time => Date.parse(time) > entered).length;
return Math.floor((now - entered) / (7 * 86_400_000));
}
/**
* Static CASE_QUARANTINE problems, shared by the free policy test and the
* weekly gate: an id that is not a blocking-tier E2E case, a missing field,
* a failure class outside detector / harness / model-latency (a product
* defect is fixed or named, never quarantined), a malformed or future date,
* and a tier over its cap.
*/
export function quarantinePolicyProblems(quarantine: Record<string, QuarantineEntry>,
registry: Registry = LIVE_REGISTRY, policy: PassRatePolicy = EVAL_POLICY, now = Date.now()): Alarm[] {
const problems: Alarm[] = [];
const invalid = (id: string, message: string) => problems.push({ kind: 'quarantine-invalid', case: id, message: `${id}: ${message}` });
const perTier = new Map<string, number>();
for (const [id, entry] of Object.entries(quarantine)) {
const tier = registry.tiers[id];
if (!tier || !(id in registry.kinds)) { invalid(id, 'CASE_QUARANTINE names no registered E2E case'); continue; }
if (!BLOCKING_TIERS.includes(tier)) invalid(id, `tier ${tier} is not blocking; only ${BLOCKING_TIERS.join(' and ')} cases are quarantined`);
for (const field of ['reason', 'failureClass', 'tracking', 'owner', 'enteredAt', 'exit'] as const) {
if (typeof entry[field] !== 'string' || !entry[field].trim()) invalid(id, `missing ${field}`);
}
if (typeof entry.reason === 'string' && entry.reason.trim().length < 40) invalid(id, 'reason must be a written diagnosis (at least 40 characters)');
if (!QUARANTINE_FAILURE_CLASSES.includes(entry.failureClass)) {
invalid(id, `failureClass ${JSON.stringify(entry.failureClass)} is not ${QUARANTINE_FAILURE_CLASSES.join(', ')}; a product defect is fixed or named as a red, never quarantined`);
}
const entered = Date.parse(entry.enteredAt);
if (!/^\d{4}-\d{2}-\d{2}$/.test(entry.enteredAt ?? '') || !Number.isFinite(entered)) invalid(id, 'enteredAt must be YYYY-MM-DD');
else if (entered > now) invalid(id, 'enteredAt is in the future');
perTier.set(tier, (perTier.get(tier) ?? 0) + 1);
}
for (const [tier, count] of perTier) {
const size = Object.values(registry.tiers).filter(value => value === tier).length;
const cap = Math.floor(size * policy.quarantine.capFraction);
if (count > cap) problems.push({ kind: 'quarantine-cap', case: tier,
message: `${count} quarantined ${tier} cases exceed the ${pct(policy.quarantine.capFraction)} cap (${cap} of ${size})` });
}
return problems;
}
export function analyzePassRates(records: TrialRecord[], options: AnalyzeOptions = {}): PassRateReport {
const registry = options.registry ?? LIVE_REGISTRY;
const quarantine = options.quarantine ?? CASE_QUARANTINE;
const policy = options.policy ?? EVAL_POLICY;
const now = options.now ?? Date.now();
const byCase = new Map<string, TrialRecord[]>();
for (const record of records) {
const list = byCase.get(record.case) ?? [];
list.push(record);
byCase.set(record.case, list);
}
for (const id of options.manualReviews ?? []) if (!byCase.has(id)) byCase.set(id, []);
const cases: CasePassRate[] = [];
for (const [id, list] of [...byCase].sort(([a], [b]) => a.localeCompare(b))) {
list.sort((a, b) => at(a).localeCompare(at(b)) || runOf(a).localeCompare(runOf(b)) || a.trial - b.trial);
const groups = new Map<string, TrialRecord[]>();
for (const record of list) {
const key = record.policy_version === 0 ? 'pre-policy'
: [record.series_identity ?? 'unknown', record.model ?? 'unknown', record.cli_version ?? 'unknown', `v${record.policy_version}`].join('|');
const group = groups.get(key) ?? [];
group.push(record);
groups.set(key, group);
}
const series = [...groups].map(([key, group]) => seriesStats(key, group))
.sort((a, b) => a.lastSeen.localeCompare(b.lastSeen));
const post = series.filter(entry => entry.policyVersion !== 0);
const current = post[post.length - 1] ?? null;
const previous = post[post.length - 2] ?? null;
const scored = current ? groups.get(current.key)!.filter(record => record.outcome !== 'skipped') : [];
const latestRun = scored.length ? runOf(scored[scored.length - 1]!) : null;
const latest = scored.filter(record => runOf(record) === latestRun);
const prior = scored.filter(record => runOf(record) !== latestRun);
const priorPasses = prior.filter(record => record.outcome === 'passed').length;
const entryRate = policy.quarantine.entry.rate;
let label: PassRateLabel;
if (latest.length > 0 && latest.every(record => record.outcome === 'failed')
&& prior.length > 0 && wilsonInterval(priorPasses, prior.length).lo >= entryRate) label = 'BROKEN';
else if (!current || current.trials < policy.quarantine.entry.minTrials) label = 'INCONCLUSIVE';
else if (current.interval.hi < entryRate) label = 'FAILING';
else if (current.passes < current.trials && current.interval.lo < entryRate) label = 'FLAKY';
else label = 'PASSING';
cases.push({
case: id, kind: registry.kinds[id] ?? list[0]!.kind, tier: caseTier(id, registry),
quarantined: id in quarantine, label, current, previous,
manualReviews: (options.manualReviews ?? []).filter(name => name === id).length,
prePolicy: series.find(entry => entry.policyVersion === 0) ?? null, series,
latestRun: latestRun ? { runId: latestRun, passes: latest.filter(record => record.outcome === 'passed').length, trials: latest.length } : null,
});
}
const alarms: Alarm[] = [];
const rate = (stats: SeriesStats) => stats.passes / stats.trials;
for (const entry of cases) {
const current = entry.current;
if (!current) continue;
const below = current.trials >= policy.quarantine.entry.minTrials && rate(current) < policy.quarantine.entry.rate;
if (below && !entry.quarantined && BLOCKING_TIERS.includes(entry.tier)) alarms.push({ kind: 'drift', case: entry.case,
message: `${entry.case} passes ${current.passes}/${current.trials} (below ${pct(policy.quarantine.entry.rate)} over >= ${policy.quarantine.entry.minTrials} trials): fix it, or propose a CASE_QUARANTINE entry with a written diagnosis (product defects are never quarantined)` });
if (below && entry.kind === 'rule') alarms.push({ kind: 'rule-as-behavior', case: entry.case,
message: `${entry.case}: rule case behaving like behavior (${current.passes}/${current.trials}): fix or reclassify` });
if (entry.quarantined && current.trials >= policy.quarantine.exit.minTrials && rate(current) >= policy.quarantine.exit.rate) {
alarms.push({ kind: 'quarantine-exit', case: entry.case,
message: `${entry.case} passes ${current.passes}/${current.trials} (>= ${pct(policy.quarantine.exit.rate)}): remove its CASE_QUARANTINE entry` });
}
}
const tested = cases.filter(entry => BLOCKING_TIERS.includes(entry.tier) && entry.current && entry.previous
&& entry.current.trials >= policy.drift.fisherMinPerSide && entry.previous.trials >= policy.drift.fisherMinPerSide);
const pValues = tested.map(entry => fisherOneSidedLower(entry.current!.passes, entry.current!.trials, entry.previous!.passes, entry.previous!.trials));
for (const index of holmRejections(pValues, policy.drift.fisherAlpha)) {
const entry = tested[index]!;
alarms.push({ kind: 'regression', case: entry.case,
message: `${entry.case}: current identity ${entry.current!.passes}/${entry.current!.trials} is significantly below the previous ${entry.previous!.passes}/${entry.previous!.trials} (one-sided Fisher p=${pValues[index]!.toFixed(4)}, Holm over ${tested.length} cases)` });
}
for (const [id, entry] of Object.entries(quarantine)) {
const runs = quarantineRunsSince(entry.enteredAt, options.weeklyRuns, now);
if (runs >= policy.quarantine.expiryWeeklyRuns) alarms.push({ kind: 'quarantine-expired', case: id,
message: `${id}: entered ${runs} weekly runs ago (limit ${policy.quarantine.expiryWeeklyRuns}): fix it, name it as a red, or re-diagnose with fresh evidence` });
}
alarms.push(...quarantinePolicyProblems(quarantine, registry, policy, now));
const post = records.filter(record => record.policy_version !== 0).length;
return { policyVersion: policy.version, cases, alarms, postPolicyTrials: post, prePolicyTrials: records.length - post,
unattributed: options.unattributed ?? [], errors: options.errors ?? [] };
}
function pct(value: number): string { return `${Math.round(value * 1000) / 10}%`; }
function formatStats(stats: SeriesStats | null): string {
if (!stats) return '-';
return `${stats.passes}/${stats.trials} [${pct(stats.interval.lo)}–${pct(stats.interval.hi)}]${stats.infra ? ` (${stats.infra} infra)` : ''}`;
}
export function formatPassRates(report: PassRateReport, options: { caseFilter?: string } = {}): string {
const lines: string[] = [];
lines.push(`pass-rates: policy v${report.policyVersion}, ${report.postPolicyTrials} post-policy trial(s), ${report.prePolicyTrials} pre-policy (display only)`);
if (report.postPolicyTrials === 0) lines.push(' no post-policy trials yet: every series starts INCONCLUSIVE');
const cases = report.cases.filter(entry => !options.caseFilter || entry.case === options.caseFilter);
lines.push(' label kind tier current series pre-policy manual case');
for (const entry of cases) {
const group = entry.current ? ` ${entry.current.model} / ${entry.current.cli}` : '';
const reset = entry.previous ? ' (baseline reset)' : '';
lines.push(` ${entry.label.padEnd(12)} ${entry.kind.padEnd(8)} ${entry.tier.padEnd(8)} ${formatStats(entry.current).padEnd(29)} `
+ `${formatStats(entry.prePolicy).padEnd(18)} ${String(entry.manualReviews).padStart(6)} ${entry.case}${entry.quarantined ? ' [quarantined]' : ''}${group}${reset}`);
}
if (report.unattributed.length) lines.push(` unattributed records (${report.unattributed.length}, never guessed): ${report.unattributed.slice(0, 20).join(', ')}`);
if (report.errors.length) lines.push(` rejected ${report.errors.length} invalid trial line(s): ${report.errors.slice(0, 5).join('; ')}`);
if (report.alarms.length) {
lines.push(`ACTION REQUIRED (${report.alarms.length}):`);
for (const alarm of report.alarms) lines.push(` [${alarm.kind}] ${alarm.message}`);
}
return lines.join('\n');
}
// --- GitHub history ---
export interface WeeklyRun { id: number; attempt: number; sha: string; branch: string; createdAt: string }
export interface RunArtifact { id: number; name: string; size: number }
/** The GitHub calls pass-rates makes; injectable so the free tests never touch the network. */
export interface HistoryFetcher {
listRuns(repo: string, workflow: string, branch: string, limit: number): WeeklyRun[];
listArtifacts(repo: string, runId: number): RunArtifact[];
downloadZip(repo: string, artifactId: number, destination: string): void;
}
function gh(args: string[]): Buffer {
const result = spawnSync('gh', args, { timeout: 300_000, maxBuffer: 256 * 1024 * 1024 });
if (result.status !== 0) throw new Error(`gh ${args.slice(0, 2).join(' ')} failed: ${String(result.stderr || result.error || '').trim()}`);
return result.stdout;
}
const jsonLines = <T>(buffer: Buffer): T[] => buffer.toString('utf8').split('\n').filter(Boolean).map(line => JSON.parse(line) as T);
export const GH_HISTORY: HistoryFetcher = {
listRuns: (repo, workflow, branch, limit) => jsonLines<WeeklyRun>(gh(['api',
`repos/${repo}/actions/workflows/${workflow}/runs?branch=${encodeURIComponent(branch)}&status=completed&per_page=${limit}`,
'--jq', '.workflow_runs[] | {id, attempt: .run_attempt, sha: .head_sha, branch: .head_branch, createdAt: .created_at}'])),
listArtifacts: (repo, runId) => jsonLines<RunArtifact>(gh(['api', `repos/${repo}/actions/runs/${runId}/artifacts?per_page=100`,
'--paginate', '--jq', '.artifacts[] | select(.expired | not) | {id, name, size: .size_in_bytes}'])),
downloadZip: (repo, artifactId, destination) => fs.writeFileSync(destination, gh(['api', `repos/${repo}/actions/artifacts/${artifactId}/zip`])),
};
/** The last `limit` completed runs of `workflow` on each branch, newest first, deduplicated. */
export function listWeeklyRuns(opts: { repo: string; workflow: string; branches: string[]; limit: number; fetcher?: HistoryFetcher }): WeeklyRun[] {
const fetcher = opts.fetcher ?? GH_HISTORY;
const runs = new Map<number, WeeklyRun>();
for (const branch of opts.branches) for (const run of fetcher.listRuns(opts.repo, opts.workflow, branch, opts.limit)) runs.set(run.id, run);
return [...runs.values()].sort((a, b) => b.createdAt.localeCompare(a.createdAt));
}
/**
* Download the artifacts of one run whose names match into a per-run cache
* directory (reused on later calls) and return the extracted directories.
* Oversized or oddly named artifacts are skipped: downloads are data only.
*/
export function downloadRunArtifacts(opts: { repo: string; run: WeeklyRun; match: (name: string) => boolean; cacheDir: string;
fetcher?: HistoryFetcher; maxBytes?: number }): string[] {
const fetcher = opts.fetcher ?? GH_HISTORY;
const dirs: string[] = [];
for (const artifact of fetcher.listArtifacts(opts.repo, opts.run.id)) {
if (!opts.match(artifact.name) || !/^[A-Za-z0-9._-]+$/.test(artifact.name)) continue;
if (artifact.size > (opts.maxBytes ?? TRIAL_OUTCOMES_MAX_BYTES)) continue;
const dir = path.join(opts.cacheDir, `${opts.run.id}`, artifact.name);
if (!fs.existsSync(path.join(dir, '.complete'))) {
fs.rmSync(dir, { recursive: true, force: true });
fs.mkdirSync(dir, { recursive: true });
const zip = path.join(dir, 'artifact.zip');
fetcher.downloadZip(opts.repo, artifact.id, zip);
const unzip = spawnSync('unzip', ['-o', '-q', zip, '-d', dir], { timeout: 120_000 });
if (unzip.status !== 0) throw new Error(`unzip failed for ${artifact.name}: ${String(unzip.stderr || unzip.error || '')}`);
fs.rmSync(zip, { force: true });
fs.writeFileSync(path.join(dir, '.complete'), '');
}
dirs.push(dir);
}
return dirs;
}
function gitOutput(args: string[]): string | null {
const result = spawnSync('git', args, { encoding: 'utf8', timeout: 5_000 });
return result.status === 0 ? result.stdout.trim() : null;
}
function repoSlug(): string {
const url = gitOutput(['remote', 'get-url', 'origin']) ?? '';
return url.match(/[:/]([^/:]+\/[^/]+?)(?:\.git)?$/)?.[1] ?? 'garrytan/gstack';
}
if (import.meta.main) {
const argv = process.argv.slice(2);
const dirFlag = argv.indexOf('--dir');
const dir = dirFlag !== -1 ? argv[dirFlag + 1] : getProjectEvalDir();
const flag = (name: string) => { const index = argv.indexOf(name); return index === -1 ? undefined : argv[index + 1]; };
const dirs = argv.flatMap((arg, index) => arg === '--dir' && argv[index + 1] ? [argv[index + 1]!] : []);
const asJson = argv.includes('--json');
const sinceFlag = argv.indexOf('--since-days');
const sinceDays = sinceFlag !== -1 ? Number(argv[sinceFlag + 1]) || 60 : 60;
const gate = argv.includes('--gate');
const backfill = argv.includes('--backfill');
const caseFilter = flag('--case');
const runsLimit = Number(flag('--runs')) || 10;
const sinceDays = Number(flag('--since-days')) || 60;
const repo = flag('--repo') ?? repoSlug();
const workflow = flag('--workflow') ?? 'evals-periodic.yml';
const branch = flag('--branch') ?? gitOutput(['rev-parse', '--abbrev-ref', 'HEAD']) ?? 'main';
const files = collectEvalFiles(dir, sinceDays);
const series = [...aggregate(files).values()]
.sort((a, b) => b.retriedPasses - a.retriedPasses || (b.fails / Math.max(1, b.runs)) - (a.fails / Math.max(1, a.runs)));
const ledger = readFreeLedger();
const records: TrialRecord[] = [];
const unattributed = new Set<string>();
const errors: string[] = [];
let historyError: string | null = null;
let weeklyRuns: string[] | undefined;
if (asJson) {
console.log(JSON.stringify({ dir, runsScanned: files.length, tests: series, freeLedger: ledger }, null, 2));
const manualReviews: string[] = [];
const importDir = (dir: string, run: { run_id: string; sha?: string; timestamp?: string } | undefined, legacyDays: number) => {
const trials = readTrialOutcomeDir(dir);
records.push(...trials.records);
errors.push(...trials.errors);
const legacy = backfillEvalFiles(collectEvalFiles(dir, legacyDays), run);
records.push(...legacy.records);
manualReviews.push(...legacy.manualReviews);
legacy.unattributed.forEach(name => unattributed.add(name));
};
if (dirs.length) {
for (const dir of dirs) importDir(dir, undefined, sinceDays);
} else {
console.log(`flake-rank: ${files.length} finalized run file(s) under ${dir}`);
const flaky = series.filter((s) => s.retriedPasses > 0 || s.fails > 0 || s.manualAccepted > 0);
if (flaky.length === 0) {
console.log(' no retried passes and no failures recorded — clean series');
} else {
console.log(' retries fails/runs manual avg-dur test');
for (const s of flaky.slice(0, 30)) {
console.log(` ${String(s.retriedPasses).padStart(7)} ${String(s.fails).padStart(5)}/${String(s.runs).padEnd(6)} `
+ `${String(s.manualAccepted).padStart(6)} ${Math.round(s.totalDurationMs / s.totalAttempts / 1000).toString().padStart(5)}s ${s.name}`);
try {
const runs = listWeeklyRuns({ repo, workflow, branches: [...new Set([branch, 'main'])], limit: runsLimit });
weeklyRuns = runs.map(run => run.createdAt);
const cacheDir = path.join(os.homedir(), '.gstack', 'eval-pass-rates-cache', repo.replace('/', '-'));
const match = backfill
? (name: string) => name.startsWith('trial-outcomes') || /^(paid-slice-\d+|gate-census-\d+)$/.test(name)
: (name: string) => name.startsWith('trial-outcomes');
for (const run of runs) {
const dirsForRun = downloadRunArtifacts({ repo, run, match, cacheDir, maxBytes: backfill ? 64 * 1024 * 1024 : undefined });
for (const dir of dirsForRun) importDir(dir, { run_id: `${run.id}`, sha: run.sha, timestamp: run.createdAt }, 3650);
}
} catch (error) {
historyError = error instanceof Error ? error.message : String(error);
}
}
const report = analyzePassRates(records, { weeklyRuns, unattributed: [...unattributed].sort(), errors, manualReviews });
const ledger = readFreeLedger();
if (asJson) {
console.log(JSON.stringify({ repo, workflow, branch, dirs, historyError, ...report, freeLedger: ledger }, null, 2));
} else {
if (historyError) console.log(`pass-rates: history unavailable (${historyError}); every label below is INCONCLUSIVE`);
console.log(formatPassRates(report, { caseFilter }));
if (ledger.length > 0) {
const byFile = new Map<string, number>();
for (const e of ledger) byFile.set(e.file, (byFile.get(e.file) ?? 0) + 1);
@@ -147,4 +701,5 @@ if (import.meta.main) {
}
}
}
if (gate && (historyError || report.alarms.length)) process.exit(1);
}
+35
View File
@@ -0,0 +1,35 @@
#!/usr/bin/env bun
/**
* Stamp `series_identity` on a report's trial-outcomes JSONL (the pass-rates
* history key: a hash of each case's own touchfiles, GLOBAL_TOUCHFILES
* excluded; scripts/eval-flake-rank.ts caseSeriesIdentities). A separate step
* after `test-paid-shards.ts --report`, so the paid runner's closure never
* imports the history tool.
*
* Usage: bun run scripts/eval-trial-series.ts <trial-outcomes.jsonl>
*/
import * as fs from 'node:fs';
import * as path from 'node:path';
import { caseSeriesIdentities } from './eval-flake-rank';
import { formatTrialOutcomes, parseTrialOutcomes } from '../test/helpers/eval-store';
const ROOT = path.resolve(import.meta.dir, '..');
/** Rewrite the file with every record stamped; an invalid line fails the whole stamp. */
export function stampTrialSeries(file: string, root = ROOT): number {
const { records, errors } = parseTrialOutcomes(fs.readFileSync(file, 'utf8'));
if (errors.length) throw new Error(`${file}: ${errors.join('; ')}`);
const identities = caseSeriesIdentities([...new Set(records.map(record => record.case))], root);
const stamped = records.map(record => ({ ...record, series_identity: identities[record.case] }));
fs.writeFileSync(file, formatTrialOutcomes(stamped));
return stamped.length;
}
if (import.meta.main) {
const file = process.argv[2];
if (!file) {
console.error('usage: bun run scripts/eval-trial-series.ts <trial-outcomes.jsonl>');
process.exit(2);
}
console.log(`[eval-trial-series] stamped ${stampTrialSeries(file)} record(s) in ${file}`);
}
+147 -136
View File
@@ -1,190 +1,201 @@
{
"version": 2,
"recordedAt": "2026-09-28T09:05:00Z",
"source": "periodic census run 36385945043 (eval-slices = periodic, gate-census = gate); timed-out shards record their wall, self-skipping shards 1s; case shards (<file>#<case>) from the Bun per-case walls in that run (gate census log) and the eval JSON slowest cases (periodic)",
"recordedAt": "2026-09-29T19:29:33.835Z",
"tiers": {
"gate": {
"test/llm-judge-recommendation.test.ts": 1000,
"test/skill-e2e-ask-user-question-format-compliance.test.ts": 60000,
"test/skill-e2e-ask-user-question-format-compliance.test.ts": 63904,
"test/skill-e2e-autoplan-dual-voice.test.ts": 1000,
"test/skill-e2e-bws.test.ts": 82000,
"test/skill-e2e-bws.test.ts": 101092,
"test/skill-e2e-context-skills.test.ts": 1000,
"test/skill-e2e-coverage-audit.test.ts": 46000,
"test/skill-e2e-cso.test.ts": 251000,
"test/skill-e2e-deploy.test.ts": 427000,
"test/skill-e2e-coverage-audit.test.ts": 50849,
"test/skill-e2e-cso.test.ts": 227394,
"test/skill-e2e-deploy.test.ts": 423978,
"test/skill-e2e-design.test.ts": 261000,
"test/skill-e2e-design.test.ts#design-review-detector-shim": 51000,
"test/skill-e2e-design.test.ts#design-review-detector-shim-dom": 108000,
"test/skill-e2e-design.test.ts#design-review-plugin-handoff": 110000,
"test/skill-e2e-design.test.ts#plan-design-review-no-ui-scope": 43000,
"test/skill-e2e-diagram.test.ts": 32000,
"test/skill-e2e-docsync-spawned.test.ts": 51000,
"test/skill-e2e-design.test.ts#design-review-detector-shim": 42811,
"test/skill-e2e-design.test.ts#design-review-detector-shim-dom": 81914,
"test/skill-e2e-design.test.ts#design-review-plugin-handoff": 126642,
"test/skill-e2e-design.test.ts#plan-design-review-no-ui-scope": 24431,
"test/skill-e2e-diagram.test.ts": 29998,
"test/skill-e2e-docsync-spawned.test.ts": 137929,
"test/skill-e2e-first-task-scaffold.test.ts": 1000,
"test/skill-e2e-gbrain-roundtrip-local.test.ts": 1000,
"test/skill-e2e-hermetic-canary.test.ts": 8000,
"test/skill-e2e-investigate-owned-completion.test.ts": 45000,
"test/skill-e2e-investigate-owned-termination.test.ts": 44000,
"test/skill-e2e-hermetic-canary.test.ts": 8813,
"test/skill-e2e-investigate-owned-completion.test.ts": 45178,
"test/skill-e2e-investigate-owned-termination.test.ts": 48608,
"test/skill-e2e-ios-device.test.ts": 1000,
"test/skill-e2e-learnings.test.ts": 35000,
"test/skill-e2e-office-hours-auto-mode.test.ts": 67000,
"test/skill-e2e-learnings.test.ts": 28899,
"test/skill-e2e-office-hours-auto-mode.test.ts": 83101,
"test/skill-e2e-office-hours-brain-writeback.test.ts": 1000,
"test/skill-e2e-office-hours-phase4.test.ts": 1000,
"test/skill-e2e-office-hours.test.ts": 1000,
"test/skill-e2e-plan-ceo-finding-floor.test.ts": 516000,
"test/skill-e2e-plan-ceo-plan-mode.test.ts": 39000,
"test/skill-e2e-plan-ceo-finding-floor.test.ts": 297738,
"test/skill-e2e-plan-ceo-plan-mode.test.ts": 36812,
"test/skill-e2e-plan-decision-classification.test.ts": 1000,
"test/skill-e2e-plan-design-with-ui.test.ts": 500000,
"test/skill-e2e-plan-devex-finding-floor.test.ts": 233000,
"test/skill-e2e-plan-design-with-ui.test.ts": 595151,
"test/skill-e2e-plan-devex-finding-floor.test.ts": 207635,
"test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 1000,
"test/skill-e2e-plan-devex-plan-mode.test.ts": 85000,
"test/skill-e2e-plan-devex-plan-mode.test.ts": 115679,
"test/skill-e2e-plan-format.test.ts": 1000,
"test/skill-e2e-plan-mode-no-op.test.ts": 206000,
"test/skill-e2e-plan-mode-no-op.test.ts": 217184,
"test/skill-e2e-plan-prosons.test.ts": 1000,
"test/skill-e2e-plan-tune.test.ts": 60000,
"test/skill-e2e-plan-tune.test.ts": 63925,
"test/skill-e2e-plan.test.ts": 251000,
"test/skill-e2e-plan.test.ts#codex-offered-ceo-review": 55000,
"test/skill-e2e-plan.test.ts#codex-offered-design-review": 59000,
"test/skill-e2e-plan.test.ts#codex-offered-eng-review": 58000,
"test/skill-e2e-plan.test.ts#codex-offered-office-hours": 54000,
"test/skill-e2e-plan.test.ts#office-hours-spec-review": 34000,
"test/skill-e2e-plan.test.ts#plan-ceo-review-benefits": 52000,
"test/skill-e2e-plan.test.ts#plan-review-report": 51000,
"test/skill-e2e-plan.test.ts#codex-offered-ceo-review": 49880,
"test/skill-e2e-plan.test.ts#codex-offered-design-review": 54895,
"test/skill-e2e-plan.test.ts#codex-offered-eng-review": 59818,
"test/skill-e2e-plan.test.ts#codex-offered-office-hours": 50442,
"test/skill-e2e-plan.test.ts#office-hours-spec-review": 33866,
"test/skill-e2e-plan.test.ts#plan-ceo-review-benefits": 48811,
"test/skill-e2e-plan.test.ts#plan-review-report": 76406,
"test/skill-e2e-qa-bugs.test.ts": 1000,
"test/skill-e2e-qa-workflow.test.ts": 398000,
"test/skill-e2e-retro.test.ts": 152000,
"test/skill-e2e-qa-callers.test.ts": 737164,
"test/skill-e2e-qa-functional-fix.test.ts": 235133,
"test/skill-e2e-qa-functional.test.ts": 356641,
"test/skill-e2e-qa-workflow.test.ts": 427303,
"test/skill-e2e-retro.test.ts": 141804,
"test/skill-e2e-review-army.test.ts": 520000,
"test/skill-e2e-review-army.test.ts#review-army-delivery-audit": 48000,
"test/skill-e2e-review-army.test.ts#review-army-json-findings": 24000,
"test/skill-e2e-review-army.test.ts#review-army-migration-safety": 140000,
"test/skill-e2e-review-army.test.ts#review-army-perf-n-plus-one": 244000,
"test/skill-e2e-review-army.test.ts#review-army-quality-score": 64000,
"test/skill-e2e-review-attribution.test.ts": 87000,
"test/skill-e2e-review.test.ts": 129000,
"test/skill-e2e-session-intelligence.test.ts": 59000,
"test/skill-e2e-review-army.test.ts#review-army-delivery-audit": 50219,
"test/skill-e2e-review-army.test.ts#review-army-json-findings": 21332,
"test/skill-e2e-review-army.test.ts#review-army-migration-safety": 145112,
"test/skill-e2e-review-army.test.ts#review-army-perf-n-plus-one": 241299,
"test/skill-e2e-review-army.test.ts#review-army-quality-score": 86836,
"test/skill-e2e-review-attribution.test.ts": 77592,
"test/skill-e2e-review.test.ts": 136132,
"test/skill-e2e-session-intelligence.test.ts": 55332,
"test/skill-e2e-shared-libs-paths.test.ts": 680000,
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-index-flags": 201000,
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-path-eligibility": 252000,
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-prior-coverage": 227000,
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-index-flags": 190202,
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-path-eligibility": 205615,
"test/skill-e2e-shared-libs-paths.test.ts#shared-libs-review-prior-coverage": 214728,
"test/skill-e2e-shared-libs.test.ts": 658000,
"test/skill-e2e-shared-libs.test.ts#shared-libs-read-only": 230000,
"test/skill-e2e-shared-libs.test.ts#shared-libs-review-lifecycle": 278000,
"test/skill-e2e-shared-libs.test.ts#shared-libs-review-revalidation": 427000,
"test/skill-e2e-shared-libs.test.ts#shared-libs-unsupported-git": 168000,
"test/skill-e2e-shared-libs.test.ts#shared-libs-read-only": 197454,
"test/skill-e2e-shared-libs.test.ts#shared-libs-review-lifecycle": 267080,
"test/skill-e2e-shared-libs.test.ts#shared-libs-review-revalidation": 316752,
"test/skill-e2e-shared-libs.test.ts#shared-libs-unsupported-git": 170776,
"test/skill-e2e-ship-docsync.test.ts": 129000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-completion": 491000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-current": 364000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-failure": 244000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-late-result": 169000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-launch-failure": 109000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-missing-asset": 173000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-missing-marker": 131000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-recovery": 245000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-stale-after": 285000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-stale-before": 210000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-store": 366000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-timeout-unsettled": 151000,
"test/skill-e2e-ship-hook-consent.test.ts": 63000,
"test/skill-e2e-ship-hook-refresh.test.ts": 70000,
"test/skill-e2e-skillify.test.ts": 188000,
"test/skill-e2e-third-party-actions.test.ts": 70000,
"test/skill-e2e-triage.test.ts": 80000,
"test/skill-e2e-workflow.test.ts": 300000,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-completion": 366116,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-current": 434135,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-failure": 164667,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-late-result": 214218,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-launch-failure": 140987,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-missing-asset": 142564,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-missing-marker": 117648,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-recovery": 225432,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-stale-after": 269572,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-stale-before": 197633,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-store": 396588,
"test/skill-e2e-ship-docsync.test.ts#ship-docsync-timeout-unsettled": 152278,
"test/skill-e2e-ship-hook-consent.test.ts": 69995,
"test/skill-e2e-ship-hook-refresh.test.ts": 70007,
"test/skill-e2e-ship-skip.test.ts": 80318,
"test/skill-e2e-skillify.test.ts": 191127,
"test/skill-e2e-third-party-actions.test.ts": 82102,
"test/skill-e2e-triage.test.ts": 54088,
"test/skill-e2e-workflow.test.ts": 184925,
"test/skill-llm-eval.test.ts": 354000,
"test/skill-routing-e2e.test.ts": 1000
},
"periodic": {
"test/carve-section-loading-browse.test.ts": 109000,
"test/carve-section-loading-codex.test.ts": 118000,
"test/carve-section-loading-design-consultation.test.ts": 349000,
"test/carve-section-loading-design-html.test.ts": 259000,
"test/carve-section-loading-design-shotgun.test.ts": 194000,
"test/carve-section-loading-document-release.test.ts": 95000,
"test/carve-section-loading-land-and-deploy.test.ts": 185000,
"test/carve-section-loading-plan-design-review.test.ts": 284000,
"test/carve-section-loading-plan-devex-review.test.ts": 396000,
"test/carve-section-loading-plan-eng-review.test.ts": 301000,
"test/carve-section-loading-qa.test.ts": 114000,
"test/carve-section-loading-retro.test.ts": 230000,
"test/carve-section-loading-review.test.ts": 174000,
"test/carve-section-loading-setup-gbrain.test.ts": 91000,
"test/carve-section-loading-spec.test.ts": 150000,
"test/carve-section-loading-browse.test.ts": 92911,
"test/carve-section-loading-codex.test.ts": 150172,
"test/carve-section-loading-design-consultation.test.ts": 318385,
"test/carve-section-loading-design-html.test.ts": 245660,
"test/carve-section-loading-design-shotgun.test.ts": 140298,
"test/carve-section-loading-document-release.test.ts": 97810,
"test/carve-section-loading-land-and-deploy.test.ts": 156293,
"test/carve-section-loading-plan-design-review.test.ts": 223606,
"test/carve-section-loading-plan-devex-review.test.ts": 384295,
"test/carve-section-loading-plan-eng-review.test.ts": 297627,
"test/carve-section-loading-qa.test.ts": 149637,
"test/carve-section-loading-retro.test.ts": 199931,
"test/carve-section-loading-review.test.ts": 247902,
"test/carve-section-loading-setup-gbrain.test.ts": 80994,
"test/carve-section-loading-spec.test.ts": 176639,
"test/codex-e2e-recommendation-substance.test.ts": 1000,
"test/codex-e2e-shared-libs.test.ts": 1000,
"test/codex-e2e-sol-scope.test.ts": 1000,
"test/codex-e2e.test.ts": 1000,
"test/llm-judge-recommendation.test.ts": 15000,
"test/skill-e2e-arm-benchmark.test.ts": 76000,
"test/llm-judge-recommendation.test.ts": 15334,
"test/skill-e2e-arm-benchmark.test.ts": 75588,
"test/skill-e2e-aside.test.ts": 1000,
"test/skill-e2e-auq-consistency.test.ts": 75000,
"test/skill-e2e-auq-matrix.test.ts": 320000,
"test/skill-e2e-auq-verbose-vs-carved-ab.test.ts": 66000,
"test/skill-e2e-auto-decide-preserved.test.ts": 113000,
"test/skill-e2e-autoplan-dual-voice.test.ts": 532000,
"test/skill-e2e-benchmark-providers.test.ts": 9000,
"test/skill-e2e-auq-consistency.test.ts": 70429,
"test/skill-e2e-auq-matrix.test.ts": 165796,
"test/skill-e2e-auq-verbose-vs-carved-ab.test.ts": 58547,
"test/skill-e2e-auto-decide-preserved.test.ts": 154931,
"test/skill-e2e-autoplan-dual-voice.test.ts": 167918,
"test/skill-e2e-benchmark-providers.test.ts": 9495,
"test/skill-e2e-bws.test.ts": 1000,
"test/skill-e2e-context-skills.test.ts": 181000,
"test/skill-e2e-context-skills.test.ts": 169497,
"test/skill-e2e-coverage-audit.test.ts": 1000,
"test/skill-e2e-cso.test.ts": 358000,
"test/skill-e2e-cso.test.ts": 253498,
"test/skill-e2e-deploy.test.ts": 1000,
"test/skill-e2e-design.test.ts": 817000,
"test/skill-e2e-design.test.ts#design-consultation-core": 204000,
"test/skill-e2e-design.test.ts#design-html-slop-gate": 205000,
"test/skill-e2e-design.test.ts#plan-design-review-plan-mode": 283000,
"test/skill-e2e-diagram.test.ts": 166000,
"test/skill-e2e-first-task-scaffold.test.ts": 9000,
"test/skill-e2e-design.test.ts#design-consultation-core": 159257,
"test/skill-e2e-design.test.ts#design-consultation-existing": 175607,
"test/skill-e2e-design.test.ts#design-consultation-preview": 114638,
"test/skill-e2e-design.test.ts#design-consultation-research": 109573,
"test/skill-e2e-design.test.ts#design-html-slop-gate": 158970,
"test/skill-e2e-design.test.ts#plan-design-review-plan-mode": 300166,
"test/skill-e2e-diagram.test.ts": 55187,
"test/skill-e2e-first-task-scaffold.test.ts": 8563,
"test/skill-e2e-gbrain-roundtrip-local.test.ts": 1000,
"test/skill-e2e-health.test.ts": 165000,
"test/skill-e2e-health.test.ts": 188584,
"test/skill-e2e-hermetic-canary.test.ts": 1000,
"test/skill-e2e-ios-device.test.ts": 1000,
"test/skill-e2e-learnings.test.ts": 1000,
"test/skill-e2e-office-hours-brain-writeback.test.ts": 174000,
"test/skill-e2e-office-hours-phase4.test.ts": 60000,
"test/skill-e2e-office-hours-brain-writeback.test.ts": 203044,
"test/skill-e2e-office-hours-design-draft.test.ts": 285868,
"test/skill-e2e-office-hours-phase4.test.ts": 41421,
"test/skill-e2e-office-hours-section-loading.test.ts": 1200000,
"test/skill-e2e-office-hours.test.ts": 119000,
"test/skill-e2e-outside-plan-disabled.test.ts": 43000,
"test/skill-e2e-office-hours.test.ts": 136970,
"test/skill-e2e-outside-plan-disabled.test.ts": 53292,
"test/skill-e2e-outside-voice.test.ts": 1000,
"test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts": 81000,
"test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts": 66000,
"test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts": 42000,
"test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts": 278000,
"test/skill-e2e-plan-ceo-mode-routing.test.ts": 575000,
"test/skill-e2e-plan-ceo-review-section-loading.test.ts": 604000,
"test/skill-e2e-plan-ceo-split-overflow.test.ts": 1332000,
"test/skill-e2e-plan-decision-classification.test.ts": 122000,
"test/skill-e2e-plan-design-finding-floor.test.ts": 175000,
"test/skill-e2e-plan-design-plan-mode.test.ts": 117000,
"test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 83000,
"test/skill-e2e-plan-eng-finding-floor.test.ts": 282000,
"test/skill-e2e-plan-eng-multi-finding-batching.test.ts": 734000,
"test/skill-e2e-plan-eng-plan-mode.test.ts": 91000,
"test/skill-e2e-plan-format.test.ts": 282000,
"test/skill-e2e-plan-prosons.test.ts": 180000,
"test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts": 85282,
"test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts": 80465,
"test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts": 49282,
"test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts": 318532,
"test/skill-e2e-plan-ceo-mode-routing.test.ts": 443504,
"test/skill-e2e-plan-ceo-review-section-loading.test.ts": 342250,
"test/skill-e2e-plan-ceo-split-overflow.test.ts": 504266,
"test/skill-e2e-plan-decision-classification.test.ts": 93469,
"test/skill-e2e-plan-design-finding-floor.test.ts": 155538,
"test/skill-e2e-plan-design-plan-mode.test.ts": 106608,
"test/skill-e2e-plan-devex-peer-comparison-classification.test.ts": 79786,
"test/skill-e2e-plan-eng-finding-floor.test.ts": 388666,
"test/skill-e2e-plan-eng-multi-finding-batching.test.ts": 1318488,
"test/skill-e2e-plan-eng-plan-mode.test.ts": 229226,
"test/skill-e2e-plan-format.test.ts": 219670,
"test/skill-e2e-plan-prosons.test.ts": 163659,
"test/skill-e2e-plan-tune.test.ts": 1000,
"test/skill-e2e-plan.test.ts": 824000,
"test/skill-e2e-plan.test.ts#plan-ceo-review": 123000,
"test/skill-e2e-plan.test.ts#plan-ceo-review-selective": 244000,
"test/skill-e2e-plan.test.ts#plan-eng-review-artifact": 153000,
"test/skill-e2e-qa-bugs.test.ts": 283000,
"test/skill-e2e-qa-workflow.test.ts": 239000,
"test/skill-e2e-retro.test.ts": 114000,
"test/skill-e2e-plan.test.ts#plan-ceo-review": 136101,
"test/skill-e2e-plan.test.ts#plan-ceo-review-expansion-energy": 91984,
"test/skill-e2e-plan.test.ts#plan-ceo-review-selective": 267212,
"test/skill-e2e-plan.test.ts#plan-eng-review": 128506,
"test/skill-e2e-plan.test.ts#plan-eng-review-artifact": 164137,
"test/skill-e2e-qa-bugs.test.ts": 333847,
"test/skill-e2e-qa-workflow.test.ts": 403953,
"test/skill-e2e-retro.test.ts": 166711,
"test/skill-e2e-review-army.test.ts": 511000,
"test/skill-e2e-review-army.test.ts#review-army-consensus": 278000,
"test/skill-e2e-review-army.test.ts#review-army-simplification": 144000,
"test/skill-e2e-review-army.test.ts#review-army-consensus": 277908,
"test/skill-e2e-review-army.test.ts#review-army-red-team": 88882,
"test/skill-e2e-review-army.test.ts#review-army-simplification": 136644,
"test/skill-e2e-review-army.test.ts#review-army-simplification-precision": 23884,
"test/skill-e2e-review-attribution.test.ts": 1000,
"test/skill-e2e-review.test.ts": 131000,
"test/skill-e2e-review.test.ts": 179362,
"test/skill-e2e-session-intelligence.test.ts": 1000,
"test/skill-e2e-setup-gbrain-bad-token.test.ts": 49000,
"test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts": 119000,
"test/skill-e2e-setup-gbrain-remote.test.ts": 110000,
"test/skill-e2e-shared-libs-periodic.test.ts": 415000,
"test/skill-e2e-ship-section-loading.test.ts": 379000,
"test/skill-e2e-skillify.test.ts": 56000,
"test/skill-e2e-sync-gbrain-readiness.test.ts": 74000,
"test/skill-e2e-setup-gbrain-bad-token.test.ts": 42912,
"test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts": 165743,
"test/skill-e2e-setup-gbrain-remote.test.ts": 121127,
"test/skill-e2e-shared-libs-periodic.test.ts": 369660,
"test/skill-e2e-ship-section-loading.test.ts": 255273,
"test/skill-e2e-skillify.test.ts": 58522,
"test/skill-e2e-sync-gbrain-readiness.test.ts": 60234,
"test/skill-e2e-third-party-actions.test.ts": 1000,
"test/skill-e2e-triage.test.ts": 1000,
"test/skill-e2e-workflow.test.ts": 1000,
"test/skill-llm-eval.test.ts": 402000,
"test/skill-routing-e2e.test.ts": 57000
"test/skill-llm-eval.test.ts": 383454,
"test/skill-routing-e2e.test.ts": 68843
}
}
}
+1087 -207
View File
File diff suppressed because it is too large. Load diff
-3
View File
@@ -36,7 +36,6 @@
"browse/test/cookie-import-transport.test.ts\tTS2741\tProperty 'preconnect' is missing in type '() => Promise<never>' but required in type 'typeof fetch'.": 1,
"browse/test/dia-gui-readiness.test.ts\tTS2352\tConversion of type '() => { status: number; stdout: string; stderr: string; }' to type '{ (command: string): SpawnSyncReturns<NonSharedBuffer>; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ status: number; stdout: string; stderr: string; }' is missing the following properties from type 'SpawnSyncReturns<NonSharedBuffer>': pid, output, signal": 1,
"browse/test/dia-launch-comparison.test.ts\tTS7016\tCould not find a declaration file for module '../../.github/scripts/dia-launch-driver.mjs'. '.github/scripts/dia-launch-driver.mjs' implicitly has an 'any' type.": 1,
"browse/test/dia-macos-qualification.test.ts\tTS2352\tConversion of type '() => { error?: undefined; status: number; stdout: string; stderr: string; } | { status: null; stdout: null; stderr: null; error: Error; }' to type '{ (command: string): SpawnSyncReturns<NonSharedBuffer>; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ error?: undefined; status: number; stdout: string; stderr: string; } | { status: null; stdout: null; stderr: null; error: Error; }' is not comparable to type 'SpawnSyncReturns<NonSharedBuffer>'. Type '{ status: null; stdout: null; stderr: null; error: Error; }' is missing the following properties from type 'SpawnSyncReturns<NonSharedBuffer>': pid, output, signal": 2,
"browse/test/dia-macos-qualification.test.ts\tTS2352\tConversion of type '() => { status: number; stdout: string; stderr: string; }' to type '{ (command: string): SpawnSyncReturns<NonSharedBuffer>; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ status: number; stdout: string; stderr: string; }' is missing the following properties from type 'SpawnSyncReturns<NonSharedBuffer>': pid, output, signal": 2,
"browse/test/dia-macos-qualification.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type 'string' is not assignable to parameter of type '\"browser_profile_unavailable\" | \"code_signing_error\" | \"debugging_pipe_unavailable\" | \"default_profile_policy\" | \"dynamic_library_error\" | \"graphics_or_bootstrap_error\" | \"keychain_access_failed\" | \"keychain_interaction_disallowed\" | \"keychain_interaction_required\"'.": 1,
"browse/test/domain-skills-e2e.test.ts\tTS2339\tProperty 'cleanup' does not exist on type 'BrowserManager'.": 1,
@@ -261,7 +260,6 @@
"test/helpers/shared-libs-eval-fixture.ts\tTS7006\tParameter 'candidate' implicitly has an 'any' type.": 2,
"test/helpers/shared-libs-path-fixture.ts\tTS2352\tConversion of type '{ root: string; repo: string; state: string; env: { GSTACK_HOME: string; }; }' to type 'SharedLibsFixture' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ root: string; repo: string; state: string; env: { GSTACK_HOME: string; }; }' is missing the following properties from type 'SharedLibsFixture': bin, trace, hookTrace, tip": 1,
"test/helpers/shared-libs-plan-actor.ts\tTS18046\t'questions' is of type 'unknown'.": 1,
"test/helpers/workflow-judge-cache.ts\tTS2352\tConversion of type 'string | number | boolean | EvalCacheValue[] | { [key: string]: EvalCacheValue; } | null' to type 'JudgeScore' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ [key: string]: EvalCacheValue; }' is missing the following properties from type 'JudgeScore': clarity, completeness, actionability, reasoning": 1,
"test/impeccable-fixtures.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type 'string | undefined' is not assignable to parameter of type 'string'. Type 'undefined' is not assignable to type 'string'.": 1,
"test/llm-judge-abort.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type '{ passed: boolean; }' is not assignable to parameter of type 'undefined'.": 3,
"test/llm-judge-frontier.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type '{ score: number; reason: string; }' is not assignable to parameter of type 'undefined'.": 1,
@@ -448,7 +446,6 @@
"test/section-capture-native-tools.test.ts\tTS2339\tProperty 'CI' does not exist on type '{ NODE_ENV?: string | undefined; TZ?: string | undefined; PATH: string; }'.": 1,
"test/session-runner-browse-errors.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type 'never[]' is not assignable to parameter of type 'undefined'.": 1,
"test/session-runner-browse-errors.test.ts\tTS7006\tParameter 'row' implicitly has an 'any' type.": 6,
"test/setup-gbrain-fixture.test.ts\tTS2352\tConversion of type '{ addTest: (row: EvalTestEntry) => number; }' to type 'EvalCollector' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ addTest: (row: EvalTestEntry) => number; }' is missing the following properties from type 'EvalCollector': tier, tests, finalized, evalDir, and 6 more.": 3,
"test/setup-gbrain-path4-caller.test.ts\tTS2339\tProperty 'diagnostic-secret' does not exist on type '{ 'no-verifier': number; 'no-install': number; 'no-init': number; 'no-registration': number; }'.": 1,
"test/setup-gbrain-path4-caller.test.ts\tTS2339\tProperty 'exit' does not exist on type '{ 'no-verifier': number; 'no-install': number; 'no-init': number; 'no-registration': number; }'.": 1,
"test/setup-gbrain-path4-caller.test.ts\tTS2339\tProperty 'leaked-claude-md' does not exist on type '{ 'no-verifier': number; 'no-install': number; 'no-init': number; 'no-registration': number; }'.": 1,
+1 -1
View File
@@ -22,7 +22,7 @@ describe('carved-skill cases each get a complete paid process budget', () => {
expect(selectPaidTestFiles(files.map(file => 'test/' + file), 'periodic').selected).toHaveLength(files.length);
expect(selectPaidTestFiles(files.map(file => 'test/' + file), 'gate').selected).toHaveLength(0);
});
test('all configured retries plus teardown fit even with within-shard concurrency one', () => {
test('every case run plus teardown fits even with within-shard concurrency one', () => {
for (const file of files) {
const attempts = retriesForFiles(['test/' + file]) + 1;
expect(CAPTURE_LONG_MS * attempts + 10_000).toBeLessThan(DEFAULT_SHARD_TIMEOUT_MS);
+42 -56
View File
@@ -21,19 +21,31 @@ test('only PR runs select the fast profile; manual and scheduled coverage stays
}
});
test('receipt transport restores only this repository and PR with no broad fallback key', () => {
const steps = paid.jobs['eval-slices'].steps;
const restore = steps.filter((s: any) => s.uses?.startsWith('actions/cache/restore@'));
const save = steps.filter((s: any) => s.uses?.startsWith('actions/cache/save@'));
test('receipt transport: the planner restores only this repository and PR, the report saves one merged store', () => {
const planner = paid.jobs['plan-slices'].steps;
const restore = planner.filter((s: any) => s.uses?.startsWith('actions/cache/restore@'));
expect(restore).toHaveLength(1);
expect(save).toHaveLength(1);
expect(restore[0].if).toBe("github.event_name == 'pull_request'");
expect(restore[0].with.path).toBe('/tmp/gstack-eval-input-cache');
expect(restore[0].with['restore-keys']).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-');
expect(save[0].with.key).toBe(restore[0].with.key);
expect(save[0].with.key).toContain('${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}');
const emit = planner.find((s: any) => s.run?.includes('--emit-plan /tmp/paid-plan/manifest.json'));
expect(emit.env.EVALS_CACHE_DIR).toBe("${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }}");
const upload = planner.find((s: any) => s.with?.name === 'paid-plan');
expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/paid-plan/manifest.json', '/tmp/paid-plan/receipts']);
// Executors never restore or save a cache of their own: every slice sees the plan's one receipt set.
const executor = paid.jobs['eval-slices'].steps;
expect(executor.filter((s: any) => s.uses?.startsWith('actions/cache/'))).toHaveLength(0);
expect(executor.find((s: any) => s.name === "Seed this slice's receipts from the plan").run).toContain('cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/');
const report = paid.jobs['slices-report'].steps;
const merge = report.find((s: any) => s.name === "Merge this run's receipts");
expect(merge.run).toContain('scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache');
const save = report.filter((s: any) => s.uses?.startsWith('actions/cache/save@'));
expect(save).toHaveLength(1);
expect(save[0].with.path).toBe('/tmp/gstack-eval-input-cache');
expect(save[0].if).toContain("steps.receipts.outputs.present == 'true'");
expect(save[0].with.key).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged');
expect(report.indexOf(save[0])).toBeGreaterThan(report.indexOf(merge));
expect(paid.jobs['eval-slices'].permissions).toEqual({ contents: 'read', packages: 'read' });
expect(paid.jobs['slices-report'].permissions).toEqual({ contents: 'read' });
expect(JSON.stringify(periodic)).not.toContain('actions/cache/');
});
@@ -43,56 +55,21 @@ test('the judge binds cache receipts to the PR and installed runtime, not the co
expect(runtime.run).toContain('sha256sum /tmp/eval-runtime-manifest.json');
const run = paid.jobs['eval-slices'].steps.find((s: any) => s.run?.includes('--plan /tmp/paid-plan/manifest.json'));
expect(run.env).toMatchObject({
EVALS_CACHE_DIR: '/tmp/gstack-eval-input-cache',
EVALS_CACHE_DIR: '/tmp/paid-slice-results/receipts',
EVALS_CACHE_REPOSITORY: '${{ github.repository }}',
EVALS_CACHE_PR: '${{ github.event.pull_request.number }}',
EVALS_CACHE_RUNTIME_ID: '${{ needs.build-image.outputs.runtime-id }}',
});
});
test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('only a new passing producer can publish the next cache snapshot', () => {
const directory = mkdtempSync(join(tmpdir(), 'ci-cache-producer-'));
const receipts = join(directory, 'receipts');
const output = join(directory, 'output');
mkdirSync(receipts);
const step = paid.jobs['eval-slices'].steps.find((s: any) => s.id === 'receipts');
const script = step.run.replaceAll('/tmp/gstack-eval-input-cache', receipts);
const run = () => {
writeFileSync(output, '');
const result = spawnSync('bash', ['-e', '-c', script], {
env: { ...process.env, GITHUB_OUTPUT: output, GITHUB_RUN_ID: '42', GITHUB_RUN_ATTEMPT: '2' },
encoding: 'utf8', timeout: 5000,
});
expect(result.status, result.stderr).toBe(0);
return readFileSync(output, 'utf8');
};
try {
expect(run()).toBe('');
writeFileSync(join(receipts, 'old.json'), JSON.stringify({ proof: { source: { runId: '41/1' } } }));
writeFileSync(join(receipts, 'corrupt.json'), '{');
expect(run()).toBe('');
writeFileSync(join(receipts, 'prior-attempt.json'), JSON.stringify({ proof: { source: { runId: '42/1' } } }));
expect(run()).toBe('');
writeFileSync(join(receipts, 'fresh.json'), JSON.stringify({ proof: { source: { runId: '42/2' } } }));
expect(run()).toBe('present=true\n');
} finally { rmSync(directory, { recursive: true, force: true }); }
});
test.skipIf(!Bun.which('jq'))('the actual comment separates reused evidence, retry outcomes and deferred coverage', () => {
test.skipIf(!Bun.which('jq'))('the actual comment shows deferred coverage and never recomputes a verdict', () => {
const comment = paid.jobs['slices-comment'].steps.find((s: any) => s.name === 'Post PR comment').run as string;
const evaluate = (filter: string, value: unknown) => {
const result = spawnSync('jq', ['-r', filter], { input: JSON.stringify(value), encoding: 'utf8', timeout: 5000 });
expect(result.status, result.stderr).toBe(0);
return result.stdout.trim();
};
const stats = comment.match(/STATS=\$\(jq -r '([^']+)'/)![1]!;
expect(evaluate(stats, { tests: [
{ name: 'retry', passed: false }, { name: 'retry', passed: true },
{ name: 'exhausted', passed: false }, { name: 'exhausted', passed: false },
{ name: 'regressed', passed: true }, { name: 'regressed', passed: false },
{ name: 'reused', passed: true, execution: 'reused' },
], flaky_retries: ['retry', 'exhausted', 'regressed'].map(name => ({ name, attempts: 2 })) })).toBe('4 2 2 3 3 1');
expect(comment).toContain("printf ' | ⚠ %s cases with multiple attempts'");
expect(comment).not.toContain('group_by(.name)');
expect(comment).not.toMatch(/flaky pass\(es\)|passed only on retry|not blocking/);
const coverage = comment.match(/COVERAGE=\$\(jq -r '([^']+)'/)![1]!;
const text = evaluate(coverage, { profile: 'pr', selection: { e2e: ['probe'], judges: ['judge'] },
@@ -107,9 +84,9 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
const job = paid.jobs['slices-comment'];
expect(job.permissions).toMatchObject({ 'pull-requests': 'write' });
expect(JSON.stringify(job.steps)).not.toMatch(/actions\/checkout|setup-bun|bun run|npm |node /);
const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict');
expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json']);
expect(job.steps.find((step: any) => step.with?.name === 'report-verdict').with.path).toBe('/tmp/verdict');
const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}');
expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json', '/tmp/paid-report/report-summary.md']);
expect(job.steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}').with.path).toBe('/tmp/verdict');
const root = mkdtempSync(join(tmpdir(), 'ci-comment-'));
const paidDir = join(root, 'paid-report');
const verdictDir = join(root, 'verdict');
@@ -123,9 +100,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
writeFileSync(join(paidDir, 'judge.json'), JSON.stringify({ total_tests: 2, tier: 'llm-judge', shard: 1,
tests: [{ name: 'manual', passed: false, manual_review: { unverified: true } },
{ name: 'reused', passed: true, execution: 'reused' }], flaky_retries: [] }));
const summary = { version: 1, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0,
const summary = { version: 2, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0,
total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 }],
totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 } };
totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 },
verdict: { verdict: 'GREEN' }, headline: ['[test:paid] VERDICT GREEN — lane gate/pr, attempt 1'], panels: [],
failures: ['⚠ case-x behavior PASS 2/3 (✓✗✓) t2: timeout at turn 3 — @\u200bsomeone said no'] };
mkdirSync(join(verdictDir, 'paid-report'));
const summaryPath = join(verdictDir, 'paid-report/collector-outcomes.json');
const script = (job.steps.find((step: any) => step.name === 'Post PR comment').run as string)
@@ -142,8 +121,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
const verified = run();
expect(verified.status, verified.stderr).toBe(0);
expect(verified.stdout).toContain('⚠ MANUAL ACCEPTED (unscored)');
expect(verified.stdout).toContain('1 automated passed / 2 final results');
expect(verified.stdout).toContain('0 failed, 1 manual accepted');
expect(verified.stdout).toContain('VERDICT GREEN — lane gate/pr, attempt 1');
expect(verified.stdout).toContain('1 executed, 1 reused** rule/judge records');
expect(verified.stdout).toContain('1 manual accepted');
expect(verified.stdout).toContain('### Failures and split verdicts');
expect(verified.stdout).toContain('PASS 2/3 (✓✗✓) t2: timeout at turn 3');
const unrelatedFailure = { ...summary, files: [{ ...summary.files[0], total: 3, failed: 1,
executed: 2, attempts: 3 }], totals: { ...summary.totals, total: 3, failed: 1,
@@ -152,16 +134,20 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f
const red = run();
expect(red.status, red.stderr).toBe(0);
expect(red.stdout).toContain('❌ FAIL');
expect(red.stdout).toContain('1 failed, 1 manual accepted');
writeFileSync(summaryPath, JSON.stringify({ ...summary, verdict: { verdict: 'RED' } }));
const redVerdict = run();
expect(redVerdict.status, redVerdict.stderr).toBe(0);
expect(redVerdict.stdout).toContain('❌ FAIL');
writeFileSync(summaryPath, JSON.stringify({ ...summary, totals: { ...summary.totals, manual_accepted: 2 } }));
const tampered = run();
expect(tampered.status, tampered.stderr).toBe(0);
expect(tampered.stdout).toContain('manual acceptance unavailable/unverified');
expect(tampered.stdout).toContain('verified report unavailable');
expect(tampered.stdout).not.toContain('⚠ MANUAL ACCEPTED (unscored)');
rmSync(summaryPath);
const absent = run();
expect(absent.status, absent.stderr).toBe(0);
expect(absent.stdout).toContain('manual acceptance unavailable/unverified');
expect(absent.stdout).toContain('verified report unavailable');
} finally { rmSync(root, { recursive: true, force: true }); }
});
+14 -6
View File
@@ -3,7 +3,7 @@ import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { buildRunManifest, collectPaidTestFiles, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards';
import { buildRunManifest, collectPaidTestFiles, shardCaseId, shardTrial, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards';
import { STRICT_RETRY_CASE_BUDGETS } from './helpers/eval-budgets';
import { approvedCookieWorkflowSource, manualReviewFixture } from './helpers/manual-judge-review-fixture';
@@ -16,6 +16,11 @@ type Job = {
permissions: Record<string, string>;
steps: Step[];
};
/** A passing trial record for an isolated trial shard (the executor's current result schema). */
const trialRecord = (entry: PaidRunManifest['entries'][number]) => entry.trial ? { trial: {
case: shardCaseId(entry.file)!, trial: shardTrial(entry.file)!, ...entry.trial, outcome: 'passed' as const, cost_usd: 0, duration_ms: 1,
} } : {};
const workflows = ['evals.yml', 'evals-periodic.yml'].map(name => ({
name,
jobs: (Bun.YAML.parse(fs.readFileSync(path.join(ROOT, '.github/workflows', name), 'utf8')) as {
@@ -112,10 +117,11 @@ describe('paid CI coordination stays off the eval image', () => {
if (name === 'evals.yml') expect(report.permissions).toEqual({ contents: 'read' });
});
test(`${name}: failure logs include the hidden spool directory without uploading the rest of the cache`, () => {
const logs = jobs['eval-slices'].steps.find(step => step.with?.name === 'paid-slice-${{ matrix.slice }}-logs');
test(`${name}: shard logs include the hidden spool directory without uploading the rest of the cache`, () => {
const logs = jobs['eval-slices'].steps.find(step => step.with?.name === 'paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}');
expect(logs?.uses).toStartWith('actions/upload-artifact@');
expect(logs?.if).toBe('failure()');
// A failed trial no longer reds its runner; its log is still the evidence.
expect(logs?.if).toBe('always()');
expect(logs?.with?.['include-hidden-files']).toBe(true);
expect(String(logs?.with?.path).trim().split('\n')).toEqual([
'/home/runner/.cache/gstack-paid-shard-*.log',
@@ -217,6 +223,7 @@ describe('dependency-free CI planner and report execution', () => {
executedTests: STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === entry.file)?.cases ?? 1,
skippedTests: 0,
...(entry.budget ? { budget: entry.budget } : {}),
...trialRecord(entry),
})),
};
fs.writeFileSync(path.join(reportDir, `slice-${sliceIndex}.json`), JSON.stringify(result));
@@ -248,7 +255,8 @@ describe('dependency-free CI planner and report execution', () => {
const red = run(['--report', reportDir], tier);
expect(red.status).toBe(1);
expect(red.stderr).toContain(`${failed.outcomes[0].files[0]}: failed`);
expect(red.stdout).toContain('3 executed, 0 reused; 1 passed, 2 failed, 0 manual accepted (unscored; no score-cache credit) (6 attempt records from 1 collectors)');
// Paid evals never retry: every record counts, a later pass never hides an earlier failure.
expect(red.stdout).toContain('6 executed, 0 reused; 2 passed, 4 failed, 0 manual accepted (unscored; no score-cache credit) (6 attempt records from 1 collectors');
expect(red.stdout).toContain('3 cases with multiple attempts this run:');
expect(red.stdout).not.toMatch(/passed only on retry|not blocking/);
@@ -274,7 +282,7 @@ describe('dependency-free CI planner and report execution', () => {
outcomes: manifest.entries.filter(entry => entry.status === 'planned').map(entry => ({
files: [entry.file], status: 'passed', exitCode: 0, elapsedMs: 1,
executedTests: STRICT_RETRY_CASE_BUDGETS.find(budget => budget.file === entry.file)?.cases ?? 1,
skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}),
skippedTests: 0, ...(entry.budget ? { budget: entry.budget } : {}), ...trialRecord(entry),
})),
};
const slicePath = path.join(reportDir, 'slice-1.json');
+7 -7
View File
@@ -11,7 +11,7 @@ import { selectTests } from './helpers/test-selection';
import { E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data';
import { selectPrProfile } from '../scripts/test-pr-profile';
import { JUDGE_MS } from './helpers/eval-budgets';
import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge';
import { COOKIE_MANUAL_REVIEW_FILE, getCookieWorkflowManualReview, isManualReviewEntry } from './helpers/cookie-workflow-manual-review';
const ROOT = resolve(import.meta.dir, '..');
@@ -70,7 +70,7 @@ function actualCookieCallback(root: string, overrides: {
const records: EvalTestEntry[] = [];
const attempts = new Map<string, { attempt: number }>();
let callback: () => Promise<void> = async () => { throw new Error('Judge callback was not registered'); };
new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', registration)(
new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', 'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS', registration)(
(_suite: string, names: string[], run: () => void) => { expect(names).toEqual([NAME]); run(); },
(name: string, run: () => Promise<void>, budget: number) => { expect(name).toBe(NAME); expect(budget).toBe(JUDGE_MS + 10_000); callback = run; },
root, buildCookieWorkflowJudgeInput, (_kind: string, explicit?: string) => explicit ?? 'fixture-model',
@@ -85,7 +85,7 @@ function actualCookieCallback(root: string, overrides: {
attempts, overrides.clock ? { now: overrides.clock } : performance,
overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout,
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS,
);
return { run: () => callback(), requests, records, attempts };
}
@@ -96,7 +96,7 @@ describe('cookie workflow judge input', () => {
approveFixture(root);
const h = actualCookieCallback(root, { judge: async () => { throw refusal(); } });
await h.run();
expect(h.requests).toHaveLength(1);
expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
expect(h.records).toHaveLength(1);
expect(h.records[0]).toMatchObject({ passed: false, execution: 'executed', exit_reason: 'provider_refusal' });
expect(isManualReviewEntry(h.records[0])).toBe(true);
@@ -161,7 +161,7 @@ describe('cookie workflow judge input', () => {
const root = fixture(); approveFixture(root);
let calls = 0;
const h = actualCookieCallback(root, { judge: async () => {
if (++calls === 1) return { ...passingScore, clarity: 1 };
if (++calls <= JUDGE_PANEL_SAMPLES) return { ...passingScore, clarity: 1 };
throw refusal();
} });
await expect(h.run()).rejects.toThrow();
@@ -275,7 +275,7 @@ describe('cookie workflow judge input', () => {
let scores = passingScore;
const h = actualCookieCallback(root, { judge: async () => scores });
await h.run();
expect(h.requests).toHaveLength(1);
expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
expect(h.requests[0].prompt).toBe(input.prompt);
expect(h.requests[0].model).toBe(COOKIE_WORKFLOW_JUDGE.model);
expect(h.requests[0].signal).toBeInstanceOf(AbortSignal);
@@ -283,7 +283,7 @@ describe('cookie workflow judge input', () => {
expect(existsSync(join(root, 'cache'))).toBe(false);
const fresh = actualCookieCallback(root);
await fresh.run();
expect(fresh.requests).toHaveLength(1);
expect(fresh.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
for (const dimension of ['clarity', 'completeness', 'actionability'] as const) {
scores = { ...COOKIE_WORKFLOW_JUDGE.thresholds, [dimension]: COOKIE_WORKFLOW_JUDGE.thresholds[dimension] - 1, reasoning: 'Synthetic failing fixture score' };
await expect(h.run()).rejects.toThrow();
+76 -2
View File
@@ -4,7 +4,8 @@ import * as os from 'node:os';
import * as path from 'node:path';
import {
e2eReuseEnvironment, e2eReuseLaneProblem, e2eShardIdentity, e2eShardInputFiles, prepareE2EShardReuse,
type E2EShardReuseRequest,
mergeReceiptDirs, readPanelReceipt, selectPlanReceipts, writeNegativeReceipt, writePanelReceipt,
type E2EShardReuseRequest, type PanelReceipt,
} from '../scripts/e2e-shard-reuse';
import { buildRunManifest, fileCaseRegistration, runPaidShard, verifySliceResults, type SliceResult } from '../scripts/test-paid-shards';
@@ -127,10 +128,13 @@ describe('E2E shard reuse through the runner', () => {
test('a failed shard never publishes a receipt', async () => {
let published = 0;
const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, env: laneEnv(), log: () => {},
reuseFor: () => ({ lookup: () => null, publish: () => { published++; } }),
reuseFor: () => ({ inputKey: 'e'.repeat(64), unchanged: () => true, lookupPanelTrial: () => null,
lookup: () => null, publish: () => { published++; } }),
commandFor: () => ({ command: process.execPath, args: ['-e', 'process.exit(1)'] }) });
expect(outcome.status).toBe('failed');
expect(published).toBe(0);
// The identity rides on the outcome so the report can store the FAIL as a negative receipt.
expect(outcome.inputKey).toBe('e'.repeat(64));
});
test('the report accepts reused results only in the fast PR profile', () => {
@@ -143,3 +147,73 @@ describe('E2E shard reuse through the runner', () => {
expect(verifySliceResults(manifest, results).problems).toContain(`${FILE}: only the fast PR profile may reuse results; this lane executes fresh`);
});
});
describe('planner-side panel reuse and negative receipts', () => {
const panelPlan = { kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined: false };
const trialRequest = (trial: number, over: Partial<E2EShardReuseRequest> = {}) => request({
key: `${FILE}#setup-deploy-workflow~t${trial}`, panel: panelPlan, ...over });
const source = (completedAt: number, runId = '1001/1') => ({ runId, revision: 'd'.repeat(40), completedAt });
const panel = (key: string, outcomes: Array<'passed' | 'failed'>, completedAt = Date.now() - 1_000): PanelReceipt => ({
schema: 1, key, case: 'setup-deploy-workflow', kind: 'behavior', panel: { n: 3, k: 2 }, source: source(completedAt),
trials: outcomes.map((outcome, i) => ({ trial: i + 1, outcome, ...(outcome === 'failed' ? { failure_class: 'timeout' as const } : {}) })),
});
test('every trial of a panel shares one identity; the panel policy is part of it', () => {
const key = (r: E2EShardReuseRequest) => { const x = e2eShardIdentity(r); if (x.status !== 'eligible') throw new Error(x.reason); return x.identity.key; };
const t1 = key(trialRequest(1));
expect(key(trialRequest(2, { env: laneEnv({ GSTACK_EVAL_TRIAL: '2' }) }))).toBe(t1);
expect(key(trialRequest(1, { panel: { ...panelPlan, quarantined: true } }))).not.toBe(t1);
expect(key(request())).not.toBe(t1);
});
test('a whole PASS panel receipt is reused per trial, a split PASS keeps its failed trial', () => {
const dir = path.join(scratch, 'panel-hit');
const env = laneEnv({ EVALS_CACHE_DIR: dir });
const reuse = prepareE2EShardReuse(trialRequest(2, { env }))!;
expect(reuse.lookupPanelTrial(2)).toBeNull();
writePanelReceipt(dir, panel(reuse.inputKey, ['passed', 'failed', 'passed']));
expect(reuse.lookupPanelTrial(2)).toMatchObject({ trial: { trial: 2, outcome: 'failed', failure_class: 'timeout' }, hit: { source: { runId: '1001/1' } } });
expect(reuse.lookupPanelTrial(1)!.trial.outcome).toBe('passed');
});
test('FAIL, partial, expired or negatively receipted panels are never reused', () => {
const dir = path.join(scratch, 'panel-miss');
const key = 'a'.repeat(64);
for (const receipt of [panel(key, ['passed', 'failed', 'failed']), panel(key, ['passed', 'passed']),
panel(key, ['passed', 'passed', 'passed'], Date.now() - 2 * 24 * 60 * 60 * 1000)]) {
writePanelReceipt(dir, receipt);
expect(readPanelReceipt(dir, key)).toBeNull();
}
writePanelReceipt(dir, panel(key, ['passed', 'passed', 'passed'], Date.now() - 5_000));
expect(readPanelReceipt(dir, key)).not.toBeNull();
writeNegativeReceipt(dir, { schema: 1, key, source: source(Date.now() - 1_000, '1002/1') });
expect(readPanelReceipt(dir, key)).toBeNull();
});
test('the planner ships one filtered set: a newer FAIL blocks an older PASS, an older FAIL does not', () => {
const from = path.join(scratch, 'select-from');
const to = path.join(scratch, 'select-to');
fs.mkdirSync(from, { recursive: true });
const [blockedKey, keptKey, panelKey] = ['1', '2', '3'].map(c => c.repeat(64));
const passReceipt = (key: string, completedAt: number) => fs.writeFileSync(path.join(from, `${key}.json`),
JSON.stringify({ schema: 1, proof: { source: source(completedAt) } }));
passReceipt(blockedKey, 1_000);
writeNegativeReceipt(from, { schema: 1, key: blockedKey, source: source(2_000, '1002/1') });
passReceipt(keptKey, 3_000);
writeNegativeReceipt(from, { schema: 1, key: keptKey, source: source(2_000, '1002/1') });
writePanelReceipt(from, panel(panelKey, ['passed', 'passed']));
const result = selectPlanReceipts(from, to);
expect(result.blocked.sort()).toEqual([`${blockedKey}.json`, `${panelKey}.panel.json`].sort());
expect(fs.readdirSync(to).sort()).toEqual([`${blockedKey}.fail.json`, `${keptKey}.fail.json`, `${keptKey}.json`].sort());
});
test('merging receipt stores keeps the newest file per name', () => {
const [a, b, out] = ['merge-a', 'merge-b', 'merge-out'].map(name => path.join(scratch, name));
const key = '4'.repeat(64);
writeNegativeReceipt(a, { schema: 1, key, source: source(5_000, '1/1') });
writeNegativeReceipt(b, { schema: 1, key, source: source(9_000, '2/1') });
expect(mergeReceiptDirs(out, [a, b, path.join(scratch, 'missing')])).toBe(2);
expect(JSON.parse(fs.readFileSync(path.join(out, `${key}.fail.json`), 'utf8')).source.runId).toBe('2/1');
expect(mergeReceiptDirs(out, [a])).toBe(0);
});
});
+11 -10
View File
@@ -1,18 +1,17 @@
import { expect, test } from 'bun:test';
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards';
import { resolvePaidShardBudget, retriesForFiles, planPaidShards, parseRunManifest, verifySliceResults, runPaidShard, buildRunManifest, paidShardWallUpperBoundMs, collectPaidTestFiles, selectPaidTestFiles, isOverlayTestFile, DEFAULT_SHARD_TIMEOUT_MS, DEFAULT_JOBS, parseCliOptions, expandCaseShards, expandTrialShards, shardFile, sliceExecutionOrder, sliceSupervisedWallMs } from '../scripts/test-paid-shards';
import { FINDING_RETRY_BUDGETS, ALL_TIERS, SHARD_RESERVE_MS } from './helpers/eval-budgets';
import fs from 'node:fs';
import os from 'node:os';
import path from 'node:path';
for (const budget of FINDING_RETRY_BUDGETS) {
test(`${budget.file}: supervision preserves every existing attempt and retry`, () => {
test(`${budget.file}: supervision covers its one run of every case`, () => {
expect(budget.testMs).toBe(1_500_000);
// A 25-minute case is past RETRY_MAX_CASE_MS: a timed-out attempt is its verdict.
expect(budget.retries).toBe(0);
expect(retriesForFiles([budget.file])).toBe(budget.retries);
// Paid evals never retry: a timed-out case is its verdict.
expect(retriesForFiles([budget.file])).toBe(0);
expect(budget.shardReserveMs).toBe(SHARD_RESERVE_MS);
expect(budget.shardMs).toBe(budget.cases * budget.testMs * (budget.retries + 1) + budget.shardReserveMs);
expect(budget.shardMs).toBe(budget.cases * budget.testMs + budget.shardReserveMs);
expect(resolvePaidShardBudget([budget.file])).toEqual({ timeoutMs: budget.shardMs, source: 'registered', policyId: budget.id });
const source = fs.readFileSync(path.join(import.meta.dir, '..', budget.file), 'utf8');
if (budget.file === 'test/skill-e2e-plan-ceo-split-overflow.test.ts') {
@@ -176,18 +175,20 @@ test('single-slice manifest retains all registered files with one allocation', (
test('current detach supervision covers the live-census floor', () => {
const floorFor = (tier: 'gate' | 'periodic') => {
// Case-sharded files contribute one shard per case, exactly as the runner plans.
const files = expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier);
// Case-sharded files contribute one shard per case and isolated cases one
// shard per trial, exactly as the runner plans.
const files = expandTrialShards(expandCaseShards(selectPaidTestFiles(collectPaidTestFiles(), tier).selected, tier), tier).keys;
const excess = files.reduce((n, file) => n + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - DEFAULT_SHARD_TIMEOUT_MS), 0);
return Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05);
};
const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8'));
const periodicTimeout = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]);
const gateTimeout = Number(pkg.scripts['eval:bg:gate'].match(/--timeout\s+(\d+)/)[1]);
expect(floorFor('gate')).toBe(26_471);
expect(floorFor('gate')).toBe(21_725);
expect(gateTimeout).toBe(49_320);
expect(gateTimeout).toBeGreaterThanOrEqual(floorFor('gate'));
expect(floorFor('periodic')).toBe(30_797);
expect(floorFor('periodic')).toBe(33_821);
expect(periodicTimeout).toBeGreaterThanOrEqual(floorFor('periodic'));
});
for (const jobs of [1, 2, 3]) test(`FIFO bound covers partial durations with ${jobs} workers`, () => {
+308 -3
View File
@@ -13,6 +13,13 @@ import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { aggregate, collectEvalFiles } from '../scripts/eval-flake-rank';
import { manualReviewFixture } from './helpers/manual-judge-review-fixture';
import {
analyzePassRates, attributeLegacyRecord, backfillEvalFiles, caseSeriesIdentities, downloadRunArtifacts, fisherOneSidedLower,
formatPassRates, holmRejections, listWeeklyRuns, quarantinePolicyProblems, quarantineRunsSince, readTrialOutcomeDir,
wilsonInterval, type HistoryFetcher, type PassRatePolicy, type QuarantineEntry, type Registry, type TrialRecord,
} from '../scripts/eval-flake-rank';
import { EVAL_POLICY } from './helpers/periodic-exclude-data';
import { TRIAL_OUTCOME_SCHEMA, formatTrialOutcomes } from './helpers/eval-store';
const entry = (name: string, passed: boolean, attempt: number) => ({
name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1,
@@ -39,9 +46,10 @@ describe('eval-flake-rank aggregate', () => {
const display = spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), '--dir', dir],
{ encoding: 'utf8', timeout: 10_000 });
expect(display.status, display.stderr).toBe(0);
expect(display.stdout).toContain('fails/runs manual');
expect(display.stdout).toContain('0/1');
expect(display.stdout).toContain(manual.name);
// pass-rates view: the prior automated pass is the one scored pre-policy
// trial; the manual acceptance is counted in its own column, never scored.
expect(display.stdout).toContain('pre-policy manual case');
expect(display.stdout).toMatch(new RegExp(`1/1 \\[[^\\]]+\\]\\s+1 ${manual.name.replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')}`));
fs.writeFileSync(path.join(dir, 'invalid-retry.json'), run([
{ ...ordinary, attempt: 1 }, { ...manual, attempt: 2 },
]));
@@ -97,3 +105,300 @@ describe('eval-flake-rank aggregate', () => {
fs.rmSync(dir, { recursive: true, force: true });
});
});
// --- pass-rates ---
const registry: Registry = {
kinds: { 'rule-a': 'rule', 'beh-b': 'behavior', 'gate-c': 'rule', 'mar-d': 'rule', 'judge one': 'judge',
...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, 'rule'])) },
tiers: { 'rule-a': 'periodic', 'beh-b': 'periodic', 'gate-c': 'gate', 'mar-d': 'marathon',
...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, i < 9 ? 'gate' : 'periodic'])) },
touchfiles: { 'rule-a': ['test/skill-e2e-a.test.ts', 'a/**'], 'beh-b': ['test/skill-e2e-b.test.ts', 'b/**'],
'gate-c': ['test/skill-e2e-shared.test.ts'], 'mar-d': ['test/skill-e2e-shared.test.ts'] },
judgeTouchfiles: { 'judge one': ['j/SKILL.md'] },
globals: ['harness/**'],
testNames: { 'gate-c': '/gate c labeled' },
};
let clock = 0;
function trial(id: string, outcome: 'passed' | 'failed' | 'skipped', extra: Partial<TrialRecord> = {}): TrialRecord {
clock += 1;
return {
schema: TRIAL_OUTCOME_SCHEMA, case: id, file: 'test/x.test.ts', tier: registry.tiers[id] ?? 'judge',
kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1, outcome,
...(outcome === 'failed' ? { failure_class: 'assertion' as const } : {}),
duration_ms: 1, cost_usd: 0, model: 'model-x', cli_version: '2.1.284', policy_version: 1, quarantined: false,
execution: 'executed', source: 'shard', run_id: `run-${clock}`, recorded_at: new Date(Date.UTC(2026, 9, 1) + clock * 60_000).toISOString(),
series_identity: 'id-1', ...extra,
};
}
const many = (id: string, passes: number, fails: number, extra: Partial<TrialRecord> = {}) =>
[...Array.from({ length: passes }, () => trial(id, 'passed', extra)), ...Array.from({ length: fails }, () => trial(id, 'failed', extra))];
const analyze = (records: TrialRecord[], quarantine: Record<string, QuarantineEntry> = {}, extra = {}) =>
analyzePassRates(records, { registry, quarantine, now: Date.UTC(2026, 9, 2), ...extra });
const qEntry = (overrides: Partial<QuarantineEntry> = {}): QuarantineEntry => ({
reason: 'Detector graded the posture wording; 7 of 10 fresh trials failed only the regex, transcripts attached.',
failureClass: 'detector', tracking: 'TODOS.md "x"', owner: 'garrytan', enteredAt: '2026-09-29',
exit: '>= 97% over >= 10 trials on the current identity', ...overrides,
});
describe('pass-rates statistics', () => {
test('Wilson bounds match the documented policy arithmetic', () => {
expect(wilsonInterval(10, 10).lo).toBeCloseTo(0.7225, 4);
expect(wilsonInterval(6, 6).lo).toBeCloseTo(0.6097, 4);
expect(wilsonInterval(125, 125).lo).toBeCloseTo(0.9702, 4);
expect(wilsonInterval(10, 10).hi).toBe(1);
expect(wilsonInterval(0, 0)).toEqual({ lo: 0, hi: 1 });
const mid = wilsonInterval(7, 10);
expect(mid.lo).toBeGreaterThan(0.39); expect(mid.hi).toBeLessThan(0.9);
});
test('one-sided Fisher exact matches a known table and is one-sided', () => {
expect(fisherOneSidedLower(4, 6, 6, 6)).toBeCloseTo(0.227272727, 8);
expect(fisherOneSidedLower(6, 6, 4, 6)).toBe(1);
expect(fisherOneSidedLower(0, 10, 10, 10)).toBeLessThan(1e-4);
});
test('Holm rejects step-down and stops at the first non-rejection', () => {
expect([...holmRejections([0.001, 0.02, 0.04], 0.05)].sort()).toEqual([0, 1, 2]);
expect([...holmRejections([0.001, 0.03, 0.04], 0.05)]).toEqual([0]);
expect([...holmRejections([0.03, 0.04], 0.05)]).toEqual([]);
expect([...holmRejections([], 0.05)]).toEqual([]);
});
});
describe('pass-rates labels', () => {
test('thin history is INCONCLUSIVE, and after this PR every series starts there', () => {
const report = analyze(many('rule-a', 9, 0));
expect(report.cases[0]).toMatchObject({ case: 'rule-a', label: 'INCONCLUSIVE' });
expect(formatPassRates(analyze([]))).toContain('every series starts INCONCLUSIVE');
});
test('PASSING, FLAKY and FAILING come from the interval against the entry rate', () => {
expect(analyze(many('rule-a', 12, 0)).cases[0]!.label).toBe('PASSING');
expect(analyze(many('beh-b', 10, 1)).cases[0]!.label).toBe('FLAKY');
expect(analyze(many('beh-b', 2, 10)).cases[0]!.label).toBe('FAILING');
});
test('BROKEN: the latest run is 0/n after a prior interval at or above the entry rate', () => {
const prior = many('beh-b', 80, 0, { run_id: 'old' });
const latest = [1, 2, 3].map(n => trial('beh-b', 'failed', { run_id: 'new', trial: n, panel: { n: 3, k: 2 } }));
expect(analyze([...prior, ...latest]).cases[0]!.label).toBe('BROKEN');
});
test('skipped trials carry no verdict; infra failures count as failed trials', () => {
const stats = analyze([...many('rule-a', 10, 0), trial('rule-a', 'skipped'),
trial('rule-a', 'failed', { failure_class: 'infra' })]).cases[0]!.current!;
expect(stats).toMatchObject({ passes: 10, trials: 11, infra: 1 });
});
test('a new identity, model or CLI starts a new series; earlier series stay visible', () => {
const report = analyze([...many('rule-a', 10, 0), ...many('rule-a', 3, 0, { series_identity: 'id-2' }),
...many('rule-a', 2, 0, { series_identity: 'id-2', cli_version: '2.1.285' })]);
const c = report.cases[0]!;
expect(c.series).toHaveLength(3);
expect(c.current).toMatchObject({ identity: 'id-2', cli: '2.1.285', trials: 2 });
expect(c.previous).toMatchObject({ identity: 'id-2', cli: '2.1.284', trials: 3 });
expect(c.label).toBe('INCONCLUSIVE');
});
});
describe('pass-rates alarms count post-policy trials of the current series only', () => {
test('backfilled pre-policy failures are displayed but never alarm', () => {
const report = analyze(many('rule-a', 2, 20, { policy_version: 0, source: 'backfill' }));
expect(report.alarms).toEqual([]);
expect(report.cases[0]!.prePolicy).toMatchObject({ passes: 2, trials: 22 });
expect(report.cases[0]!.label).toBe('INCONCLUSIVE');
});
test('drift proposes quarantine for a blocking case below the entry rule; a rule case is flagged as behaving like behavior', () => {
const kinds = analyze([...many('rule-a', 8, 2), ...many('beh-b', 8, 2), ...many('mar-d', 0, 10)]).alarms.map(a => `${a.kind}:${a.case}`);
expect([...kinds].sort()).toEqual(['drift:beh-b', 'drift:rule-a', 'rule-as-behavior:mar-d', 'rule-as-behavior:rule-a']);
expect(analyze(many('rule-a', 19, 1)).alarms).toEqual([]);
});
test('the Fisher regression alarm needs the minimum trials on both sides', () => {
const old = many('gate-c', 6, 0, { series_identity: 'old' });
const fresh = many('gate-c', 0, 6, { series_identity: 'new' });
expect(analyze([...old, ...fresh]).alarms.map(a => a.kind)).toContain('regression');
expect(analyze([...old, ...fresh.slice(0, 5)]).alarms.map(a => a.kind)).not.toContain('regression');
});
test('quarantine exit, expiry and cap', () => {
const exit = analyze(many('beh-b', 10, 0), { 'beh-b': qEntry() }).alarms.map(a => a.kind);
expect(exit).toContain('quarantine-exit');
expect(exit).not.toContain('drift');
const weekly = Array.from({ length: 8 }, (_, i) => new Date(Date.UTC(2026, 8, 30) + i * 7 * 86_400_000).toISOString());
expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly }).alarms.map(a => a.kind)).toContain('quarantine-expired');
expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly.slice(0, 7) }).alarms.map(a => a.kind)).not.toContain('quarantine-expired');
expect(quarantineRunsSince('2026-09-01', undefined, Date.UTC(2026, 9, 27))).toBe(8);
expect(quarantineRunsSince('not a date', undefined, 0)).toBe(Number.POSITIVE_INFINITY);
});
});
describe('quarantine policy', () => {
const policy: PassRatePolicy = EVAL_POLICY;
const now = Date.UTC(2026, 9, 2);
test('a valid entry has no problems', () => {
expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]);
});
test('a product defect, a missing diagnosis or field, a bad date or a non-blocking case is rejected', () => {
const problems = (quarantine: Record<string, QuarantineEntry>) => quarantinePolicyProblems(quarantine, registry, policy, now).map(p => p.message);
expect(problems({ 'beh-b': qEntry({ failureClass: 'product' as QuarantineEntry['failureClass'] }) }).join()).toContain('never quarantined');
expect(problems({ 'beh-b': qEntry({ reason: 'flaky' }) }).join()).toContain('written diagnosis');
expect(problems({ 'beh-b': qEntry({ owner: ' ' }) }).join()).toContain('missing owner');
expect(problems({ 'beh-b': qEntry({ enteredAt: '09/29/2026' }) }).join()).toContain('YYYY-MM-DD');
expect(problems({ 'beh-b': qEntry({ enteredAt: '2027-01-01' }) }).join()).toContain('future');
expect(problems({ 'mar-d': qEntry() }).join()).toContain('not blocking');
expect(problems({ 'judge one': qEntry() }).join()).toContain('no registered E2E case');
expect(problems({ ghost: qEntry() }).join()).toContain('no registered E2E case');
});
test('at most 10% of a tier may be quarantined', () => {
// 11 periodic cases in the fixture registry: the cap is 1.
expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]);
const over = quarantinePolicyProblems({ 'beh-b': qEntry(), 'rule-a': qEntry() }, registry, policy, now);
expect(over.map(p => p.kind)).toEqual(['quarantine-cap']);
expect(over[0]!.message).toContain('2 quarantined periodic cases exceed the 10% cap (1 of 11)');
});
});
describe('pass-rates inputs', () => {
test('trial-outcomes JSONL is schema-validated; invalid lines are reported, never guessed', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-'));
const valid = trial('rule-a', 'passed');
fs.mkdirSync(path.join(dir, 'nested'));
fs.writeFileSync(path.join(dir, 'nested', 'trial-outcomes.jsonl'), formatTrialOutcomes([valid]) + '{"schema":"other"}\nnot json\n');
fs.writeFileSync(path.join(dir, 'unrelated.jsonl'), formatTrialOutcomes([valid]));
const read = readTrialOutcomeDir(dir);
expect(read.records).toHaveLength(1);
expect(read.records[0]).toMatchObject({ case: 'rule-a', series_identity: 'id-1' });
expect(read.errors).toHaveLength(2);
fs.rmSync(dir, { recursive: true, force: true });
});
test('legacy records attribute by shard suffix, id, label or single-owner file, else stay unattributed', () => {
expect(attributeLegacyRecord('/anything', 'skill-e2e-b--beh-b', registry)).toBe('beh-b');
expect(attributeLegacyRecord('rule-a', 'skill-e2e-zzz', registry)).toBe('rule-a');
expect(attributeLegacyRecord('/Rule a', 'skill-e2e-zzz', registry)).toBe('rule-a');
expect(attributeLegacyRecord('/rule a extra', 'skill-e2e-zzz', registry)).toBeNull();
expect(attributeLegacyRecord('/gate c labeled', undefined, registry)).toBe('gate-c');
expect(attributeLegacyRecord('/a display name', 'skill-e2e-a', registry)).toBe('rule-a');
expect(attributeLegacyRecord('/shared display', 'skill-e2e-shared', registry)).toBeNull();
});
test('backfill keeps only first attempts, defaults a missing attempt to 1, and labels records pre-policy', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-backfill-'));
fs.writeFileSync(path.join(dir, 'run.json'), run([
{ ...entry_('rule-a', false, 1), exit_reason: 'timeout' }, entry_('rule-a', true, 2),
{ name: 'beh-b', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 },
entry_('/unknown display', true, 1),
], { shard: 'skill-e2e-zzz' }));
const { records, unattributed } = backfillEvalFiles(collectEvalFiles(dir), { run_id: '42', sha: 'abc' }, registry);
expect(records.map(r => [r.case, r.outcome, r.failure_class, r.policy_version, r.source, r.run_id]))
.toEqual([['rule-a', 'failed', 'timeout', 0, 'backfill', '42'], ['beh-b', 'passed', undefined, 0, 'backfill', '42']]);
expect(formatTrialOutcomes(records)).toContain(TRIAL_OUTCOME_SCHEMA);
expect(unattributed).toEqual(['/unknown display']);
fs.rmSync(dir, { recursive: true, force: true });
});
test('series identity follows the case\'s own touchfiles, not GLOBAL_TOUCHFILES', () => {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-series-'));
const git = (...args: string[]) => spawnSync('git', args, { cwd: root, encoding: 'utf8', timeout: 10_000 });
for (const [file, body] of [['a/x.ts', '1'], ['b/y.ts', '1'], ['harness/run.ts', '1'], ['test/skill-e2e-a.test.ts', '1']] as const) {
fs.mkdirSync(path.join(root, path.dirname(file)), { recursive: true });
fs.writeFileSync(path.join(root, file), body);
}
const snapshot = () => { expect(git('add', '-A').status).toBe(0); return caseSeriesIdentities(['rule-a', 'beh-b'], root, registry); };
expect(git('init', '-q').status).toBe(0);
const first = snapshot();
expect(first['rule-a']).not.toBe(first['beh-b']);
fs.writeFileSync(path.join(root, 'harness/run.ts'), '2');
expect(snapshot()).toEqual(first);
fs.writeFileSync(path.join(root, 'a/x.ts'), '2');
const next = snapshot();
expect(next['rule-a']).not.toBe(first['rule-a']);
expect(next['beh-b']).toBe(first['beh-b']);
fs.rmSync(root, { recursive: true, force: true });
});
});
describe('pass-rates history fetch (injected, no network)', () => {
function storedZip(files: Record<string, string>): Buffer {
const locals: Buffer[] = [], centrals: Buffer[] = [];
let offset = 0;
for (const [name, text] of Object.entries(files)) {
const data = Buffer.from(text), fileName = Buffer.from(name), crc = Bun.hash.crc32(data) >>> 0;
const local = Buffer.alloc(30); local.writeUInt32LE(0x04034b50, 0); local.writeUInt16LE(20, 4);
local.writeUInt32LE(crc, 14); local.writeUInt32LE(data.length, 18); local.writeUInt32LE(data.length, 22); local.writeUInt16LE(fileName.length, 26);
const central = Buffer.alloc(46); central.writeUInt32LE(0x02014b50, 0); central.writeUInt16LE(20, 4); central.writeUInt16LE(20, 6);
central.writeUInt32LE(crc, 16); central.writeUInt32LE(data.length, 20); central.writeUInt32LE(data.length, 24);
central.writeUInt16LE(fileName.length, 28); central.writeUInt32LE(offset, 42);
locals.push(local, fileName, data); centrals.push(central, fileName);
offset += 30 + fileName.length + data.length;
}
const size = centrals.reduce((sum, b) => sum + b.length, 0);
const end = Buffer.alloc(22); end.writeUInt32LE(0x06054b50, 0); end.writeUInt16LE(Object.keys(files).length, 8);
end.writeUInt16LE(Object.keys(files).length, 10); end.writeUInt32LE(size, 12); end.writeUInt32LE(offset, 16);
return Buffer.concat([...locals, ...centrals, end]);
}
test('lists runs per branch, deduplicated and newest first', () => {
const fetcher: HistoryFetcher = {
listRuns: (_repo, _workflow, branch) => branch === 'main'
? [{ id: 1, attempt: 1, sha: 'a', branch, createdAt: '2026-09-01T00:00:00Z' }, { id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }]
: [{ id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }, { id: 2, attempt: 2, sha: 'b', branch, createdAt: '2026-09-08T00:00:00Z' }],
listArtifacts: () => [], downloadZip: () => { throw new Error('unused'); },
};
expect(listWeeklyRuns({ repo: 'o/r', workflow: 'evals-periodic.yml', branches: ['feature', 'main'], limit: 10, fetcher }).map(r => r.id)).toEqual([3, 2, 1]);
});
test('downloads only matching, bounded artifacts once, and caches them', () => {
const cacheDir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cache-'));
const downloads: number[] = [];
const fetcher: HistoryFetcher = {
listRuns: () => [],
listArtifacts: () => [{ id: 10, name: 'trial-outcomes-gate', size: 100 }, { id: 11, name: 'paid-slice-1', size: 100 },
{ id: 12, name: 'trial-outcomes-huge', size: 10 ** 9 }, { id: 13, name: 'trial-outcomes/../escape', size: 1 }],
downloadZip: (_repo, id, destination) => { downloads.push(id); fs.writeFileSync(destination, storedZip({ 'trial-outcomes.jsonl': formatTrialOutcomes([trial('rule-a', 'passed')]) })); },
};
const options = { repo: 'o/r', run: { id: 7, attempt: 1, sha: 's', branch: 'main', createdAt: '' }, cacheDir, fetcher,
match: (name: string) => name.startsWith('trial-outcomes') };
const dirs = downloadRunArtifacts(options);
expect(downloads).toEqual([10]);
expect(dirs).toHaveLength(1);
expect(readTrialOutcomeDir(dirs[0]!).records.map(r => r.case)).toEqual(['rule-a']);
expect(downloadRunArtifacts(options)).toEqual(dirs);
expect(downloads).toEqual([10]);
fs.rmSync(cacheDir, { recursive: true, force: true });
});
});
describe('pass-rates CLI', () => {
const cli = (args: string[]) => spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), ...args],
{ encoding: 'utf8', timeout: 20_000 });
test('--dir prints per-case pass rates; --gate fails only on ACTION REQUIRED', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cli-'));
const id = 'plan-ceo-review-format-mode';
const records = Array.from({ length: 12 }, (_, i) => ({ ...trial(id, i < 11 ? 'passed' : 'failed'), kind: 'behavior' as const, tier: 'periodic' }));
fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records));
const shown = cli(['--dir', dir, '--case', id]);
expect(shown.status, shown.stderr).toBe(0);
expect(shown.stdout).toContain(`11/12 [`);
expect(shown.stdout).toMatch(new RegExp(`FLAKY\\s+behavior\\s+periodic.*${id}`));
expect(shown.stdout).toContain('ACTION REQUIRED');
expect(shown.stdout).toContain(`[drift] ${id} passes 11/12`);
expect(cli(['--dir', dir, '--gate']).status).toBe(1);
fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records.slice(0, 11)));
const clean = cli(['--dir', dir, '--gate', '--json']);
expect(clean.status, clean.stdout).toBe(0);
expect(JSON.parse(clean.stdout).cases[0]).toMatchObject({ case: id, label: 'PASSING', current: { passes: 11, trials: 11 } });
fs.rmSync(dir, { recursive: true, force: true });
});
});
function entry_(name: string, passed: boolean, attempt: number) {
return { name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1 };
}
+107
View File
@@ -0,0 +1,107 @@
/**
* Eval kind registry (E2E_KINDS / BEHAVIOR_WHY in touchfiles-data.ts). The
* kind fixes a case's trial policy before the run, so the registry must cover
* every live case exactly once, every behavior case must name its tolerated
* deviation, and a behavior case must be isolatable as its own trial shard.
* A kind edit must re-select the case in the PR lane (map-diff).
*/
import { describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { BEHAVIOR_WHY, E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data';
import { diffTouchfileMapsCore, type TouchfileMaps } from './helpers/test-selection';
import { CASE_TEST_NAMES, fileCaseRegistration } from '../scripts/test-paid-shards';
import { isPaidTestFile } from './helpers/paid-test-set';
const ROOT = path.resolve(import.meta.dir, '..');
const KIND_RULE = "Pick the kind by what can make the verdict differ between two runs of the same commit: 'rule' when nothing "
+ "stochastic decides it or it checks a contract the product must meet every run (the default); 'behavior' when a live "
+ "model choice decides it and a sub-100% per-trial rate is acceptable (add a BEHAVIOR_WHY line); 'judge' when the only "
+ 'stochastic step is an LLM judge scoring a fixed input.';
const liveIds = [...Object.keys(E2E_TIERS), ...Object.keys(LLM_JUDGE_TOUCHFILES)];
const behaviorIds = Object.keys(E2E_KINDS).filter(id => E2E_KINDS[id] === 'behavior').sort();
describe('E2E_KINDS registry', () => {
test('every live case has exactly one kind and no kind names a dead case', () => {
const missing = liveIds.filter(id => !(id in E2E_KINDS));
expect(missing.length, missing.length ? `add to E2E_KINDS:\n${missing.map(id => ` '${id}': 'rule', // <reason>`).join('\n')}\n${KIND_RULE}` : '').toBe(0);
const unknown = Object.keys(E2E_KINDS).filter(id => !liveIds.includes(id));
expect(unknown, `E2E_KINDS names ids that are neither E2E_TIERS nor LLM_JUDGE_TOUCHFILES keys`).toEqual([]);
expect(new Set(liveIds).size).toBe(liveIds.length);
});
test('kinds are rule, behavior or judge; every LLM-judge entry is judge-kind', () => {
for (const [id, kind] of Object.entries(E2E_KINDS)) expect(['rule', 'behavior', 'judge'], id).toContain(kind);
for (const id of Object.keys(LLM_JUDGE_TOUCHFILES)) expect(E2E_KINDS[id], `${id}: a workflow judge scores a fixed input`).toBe('judge');
});
test('BEHAVIOR_WHY names the tolerance of exactly the behavior cases', () => {
expect(Object.keys(BEHAVIOR_WHY).sort()).toEqual(behaviorIds);
for (const id of behaviorIds) {
expect(BEHAVIOR_WHY[id]!.trim().length, `${id}: BEHAVIOR_WHY must say why an occasional deviation is acceptable`).toBeGreaterThanOrEqual(30);
}
});
test('a behavior case is an isolatable trial shard: known literal registration and an exact Bun test name', () => {
for (const id of behaviorIds) {
const files = E2E_TOUCHFILES[id]!.filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file));
expect(files.length, `${id}: no paid test file registers it`).toBeGreaterThan(0);
for (const file of files) {
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
expect(fileCaseRegistration(file, source).known, `${id}: ${file} has a computed registration; behavior needs a literal one`).toBe(true);
const name = CASE_TEST_NAMES[id] ?? id;
const literal = new RegExp(`\\b(?:test(?:\\.serial|\\.concurrent)?|testIfSelected|testConcurrentIfSelected)\\(\\s*(['"\`])${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\1`);
expect(literal.test(source), `${id}: ${file} must register the Bun test named '${name}'`).toBe(true);
}
}
});
test('the classification is the reviewed one: rule by default, 22 behavior, 25 judge', () => {
const counts = Object.values(E2E_KINDS).reduce<Record<string, number>>((acc, kind) => ({ ...acc, [kind]: (acc[kind] ?? 0) + 1 }), {});
expect(counts).toEqual({ rule: liveIds.length - 22 - 25, behavior: 22, judge: 25 });
// Contract-shaped cases stay rule: ask-before-decide, plan-mode no-writes,
// mandated steps, secrets, and the batching floor never ride a majority.
for (const id of ['plan-ceo-mode-routing', 'plan-eng-multi-finding-batching', 'plan-design-review-plan-mode',
'plan-eng-review-plan-mode', 'plan-ceo-section-loading', 'setup-gbrain-bad-token', 'qa-only-no-fix', 'review-sql-injection']) {
expect(E2E_KINDS[id], id).toBe('rule');
}
});
});
describe('kind edits re-select their case (map-diff)', () => {
const base = (): TouchfileMaps => ({
E2E_TOUCHFILES: { alpha: ['a/**'], beta: ['b/**'] },
E2E_TIERS: { alpha: 'gate', beta: 'periodic' },
LLM_JUDGE_TOUCHFILES: { 'judge one': ['j/SKILL.md'] },
GLOBAL_TOUCHFILES: [],
E2E_KINDS: { alpha: 'rule', beta: 'rule', 'judge one': 'judge' },
BEHAVIOR_WHY: {},
});
test('a rule -> behavior flip selects exactly that case', () => {
const next = base();
next.E2E_KINDS = { ...next.E2E_KINDS, beta: 'behavior' };
next.BEHAVIOR_WHY = { beta: 'tolerated deviation' };
expect(diffTouchfileMapsCore(base(), next).changedTests).toEqual(['beta']);
});
test('a BEHAVIOR_WHY edit alone selects its case', () => {
const old = base(); old.E2E_KINDS!.beta = 'behavior'; old.BEHAVIOR_WHY = { beta: 'one' };
const next = base(); next.E2E_KINDS!.beta = 'behavior'; next.BEHAVIOR_WHY = { beta: 'two' };
expect(diffTouchfileMapsCore(old, next).changedTests).toEqual(['beta']);
});
test('a base revision without the kind maps selects every key', () => {
const old = base(); delete old.E2E_KINDS; delete old.BEHAVIOR_WHY;
expect(diffTouchfileMapsCore(old, base()).changedTests).toEqual(['alpha', 'beta', 'judge one']);
});
test('dropping a kind entry while the case lives on counts as changed, not removed', () => {
const next = base(); delete next.E2E_KINDS!.alpha;
const result = diffTouchfileMapsCore(base(), next);
expect(result.changedTests).toEqual(['alpha']);
expect(result.removedTests).toEqual([]);
});
});
+65
View File
@@ -263,3 +263,68 @@ describe('shared setup composites (every paid lane)', () => {
}
});
});
describe('panel verdict surfaces (eval reliability policy)', () => {
type AnyJob = { if?: string; needs?: string[]; permissions?: Record<string, string>; outputs?: Record<string, string>;
strategy?: { 'max-parallel': number }; steps: Array<Step & { if?: string; uses?: string }> };
const jobsOf = (source: string) => (Bun.YAML.parse(source) as { jobs: Record<string, AnyJob> }).jobs;
test('planners size the capacity preflight with their executor cap', () => {
for (const [source, executor, manifest] of [[evalsYml, 'eval-slices', '/tmp/paid-plan/manifest.json'],
[periodicYml, 'eval-slices', '/tmp/paid-plan/manifest.json'], [periodicYml, 'gate-census', '/tmp/gate-census-plan/manifest.json']] as const) {
const jobs = jobsOf(source);
const emit = jobs['plan-slices']!.steps.find(step => step.run?.includes(`--emit-plan ${manifest} `))!;
const cap = Number(/--max-parallel (\d+)/.exec(emit.run!)?.[1]);
expect(cap, `${executor}: --max-parallel`).toBe(jobs[executor]!.strategy!['max-parallel']);
}
});
test('slice artifacts are attempt-scoped and never merged into one tree', () => {
for (const source of [evalsYml, periodicYml, marathonYml]) {
const jobs = jobsOf(source);
const uploads = Object.values(jobs).flatMap(job => job.steps).filter(step => step.uses?.startsWith('actions/upload-artifact@'))
.map(step => step.with?.name ?? '').filter(name => /slice|census-\$/.test(name));
expect(uploads.length).toBeGreaterThan(0);
for (const name of uploads) expect(name, name).toContain('-a${{ github.run_attempt }}');
const downloads = Object.values(jobs).flatMap(job => job.steps).filter(step => step.uses?.startsWith('actions/download-artifact@') && step.with?.pattern);
for (const step of downloads) expect((step.with as Record<string, unknown>)['merge-multiple'], step.with!.pattern).toBeUndefined();
}
});
test('the PR comment reads collector-outcomes v2 and never recomputes a verdict', () => {
const comment = evalsYml.slice(evalsYml.indexOf(' slices-comment:'));
expect(comment).toContain('.version == 2');
expect(comment).toContain("jq -r '.failures[]'");
expect(comment).toContain('name: report-verdict-a${{ github.run_attempt }}');
expect(evalsYml).not.toContain('group_by(.name)');
expect(comment).not.toMatch(/paid-slice-/);
const report = jobsOf(evalsYml)['slices-report']!;
expect(report.steps.some(step => step.run?.includes('scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl'))).toBe(true);
expect(report.steps.some(step => step.with?.name?.startsWith('trial-outcomes-'))).toBe(true);
});
test('the weekly report gates on pass-rate history, closes its issue on green, and re-dispatches INFRA-only reds once', () => {
const jobs = jobsOf(periodicYml);
const report = jobs.report!;
expect(report.permissions).toEqual({ contents: 'read', issues: 'write', actions: 'read' });
const gate = report.steps.find(step => step.id === 'pass-rates')!;
expect(gate.run).toContain('bun run eval:pass-rates --gate --runs 10');
expect(gate.if).toBe('always()');
for (const name of ['Upsert tracking issue on failure', 'Fail the workflow when reconciliation failed']) {
expect(report.steps.find(step => step.name === name)!.if).toContain("steps.pass-rates.outputs.exit != '0'");
}
const upsert = report.steps.find(step => step.name === 'Upsert tracking issue on failure')!;
expect(upsert.run).toContain('report-summary.md');
expect(report.steps.find(step => step.name === 'Close the tracking issue on a green run')!.run).toContain('gh issue close');
expect(report.steps.filter(step => step.with?.name?.startsWith('trial-outcomes-')).length).toBe(2);
const redispatch = jobs.redispatch!;
expect([redispatch.needs].flat()).toEqual(['report']);
expect(redispatch.permissions).toEqual({ actions: 'write' });
expect(redispatch.if).toBe("${{ !cancelled() && needs.report.outputs.redispatch == 'true' }}");
expect(redispatch.steps[0]!.run).toContain('-f redispatch_of="$GITHUB_RUN_ID"');
const classify = report.steps.find(step => step.id === 'verdict')!;
expect(classify.run).toContain('.verdict.redispatchEligible == true');
expect(classify.run).toContain('[ -z "$REDISPATCH_OF" ]');
expect(periodicYml).toMatch(/group: evals-periodic\$\{\{ inputs\.redispatch_of/);
});
});
+48 -100
View File
@@ -47,132 +47,80 @@ export const ALL_TIERS = {
export const SHARD_RESERVE_MS = 2 * 60_000;
/**
* Retry policy (approved 2026-09-29): a timed-out attempt is a verdict. Bun's
* --retry reruns a failed case after it may have spent its whole budget, so an
* automatic retry is kept only where one more attempt is short: every case of
* the file has a per-attempt budget of at most RETRY_MAX_CASE_MS, the CAPTURE
* tier plus its recording grace. Those failures are fast flake classes (API
* blips, tool hiccups) and a retry costs at most one more short attempt. Files
* with any longer case run once. Per-case budgets never change with this rule.
* Retry policy (approved 2026-09-29, eval reliability wave): paid evals never
* retry. Each case's kind (E2E_KINDS) fixes its trials before the run: `rule`
* one trial, `behavior` a panel of EVAL_POLICY.panel independent trials, and
* `judge` one case that samples its judge panel internally. A failed verdict
* is final for that run; a manual re-run adds trials under a new run attempt
* and never replaces the original verdict. Rows below keep only wall
* supervision; per-case budgets never change with this rule.
*/
export const RETRY_MAX_CASE_MS = CAPTURE_MS + 15_000;
export function retriesWithinCaseCap(caseMs: number, configuredRetries: number): number {
return caseMs <= RETRY_MAX_CASE_MS ? configuredRetries : 0;
}
/**
* Unregistered paid files that keep one automatic retry: every case budget is
* JUDGE or CAPTURE tier (test/paid-retry-supervision.test.ts scans each source).
* Registered rows below derive retries from their declared caseMs; every other
* paid file runs once.
*/
export const SHORT_CASE_RETRY_FILES: readonly string[] = [
'test/codex-e2e-sol-scope.test.ts',
'test/llm-judge-recommendation.test.ts',
'test/skill-e2e-ask-user-question-format-compliance.test.ts',
'test/skill-e2e-benchmark-providers.test.ts',
'test/skill-e2e-bws.test.ts',
'test/skill-e2e-context-skills.test.ts',
'test/skill-e2e-coverage-audit.test.ts',
'test/skill-e2e-diagram.test.ts',
'test/skill-e2e-first-task-scaffold.test.ts',
'test/skill-e2e-gbrain-roundtrip-local.test.ts',
'test/skill-e2e-hermetic-canary.test.ts',
'test/skill-e2e-investigate-owned-completion.test.ts',
'test/skill-e2e-investigate-owned-termination.test.ts',
'test/skill-e2e-learnings.test.ts',
'test/skill-e2e-plan-tune.test.ts',
'test/skill-e2e-qa-functional-fix.test.ts',
'test/skill-e2e-qa-functional.test.ts',
'test/skill-e2e-review-army.test.ts',
'test/skill-e2e-review.test.ts',
'test/skill-e2e-session-intelligence.test.ts',
'test/skill-e2e-setup-gbrain-bad-token.test.ts',
'test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts',
'test/skill-e2e-setup-gbrain-remote.test.ts',
'test/skill-e2e-ship-hook-consent.test.ts',
'test/skill-e2e-ship-hook-refresh.test.ts',
'test/skill-e2e-ship-skip.test.ts',
'test/skill-e2e-sync-gbrain-readiness.test.ts',
'test/skill-e2e-third-party-actions.test.ts',
'test/skill-e2e-triage.test.ts',
'test/skill-routing-e2e.test.ts',
];
/** Whole-file supervision covers every attempt the retry policy allows.
* These fixtures allow 25 minutes per case, so they run once.
/** Whole-file supervision for one run of every case.
* These fixtures allow 25 minutes per case.
* Reserve the sequential upper bound even when Bun runs sibling cases together.
*/
export const FINDING_RETRY_BUDGETS = [
{ file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 },
].map(({ file, cases }) => {
const retries = retriesWithinCaseCap(1_500_000, 1);
return {
file, cases,
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
testMs: 1_500_000,
caseMs: 1_500_000,
retries,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: cases * 1_500_000 * (retries + 1) + SHARD_RESERVE_MS,
};
});
].map(({ file, cases }) => ({
file, cases,
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
testMs: 1_500_000,
caseMs: 1_500_000,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: cases * 1_500_000 + SHARD_RESERVE_MS,
}));
/** Three existing captures in one 16-minute case, so the file runs once. */
/** Three existing captures in one 16-minute case. */
export const AUQ_CONSISTENCY_RETRY_BUDGET = {
file: 'test/skill-e2e-auq-consistency.test.ts',
id: 'auq-consistency-existing-retry-v1',
cases: 1,
testMs: 3 * CAPTURE_MS + 60_000,
caseMs: 3 * CAPTURE_MS + 60_000,
retries: retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1),
shardReserveMs: SHARD_RESERVE_MS,
shardMs: (3 * CAPTURE_MS + 60_000) * (retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1) + 1) + SHARD_RESERVE_MS,
shardMs: 3 * CAPTURE_MS + 60_000 + SHARD_RESERVE_MS,
} as const;
/** These fixtures have a fixed case count in every supported tier. */
export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET];
/** Whole-file walls cover all existing cases and every allowed attempt, even if
* Bun runs them sequentially. Mixed-tier files reserve their larger complete
* tier, never a currently selected subset. caseMs is the longest single case
* budget, which decides the retry (RETRY_MAX_CASE_MS). These rows add no
* case-count or model-work policy. The 10-second terms preserve the existing
* Codex/recording finalization grace.
/** Whole-file walls cover all existing cases, even if Bun runs them
* sequentially. Mixed-tier files reserve their larger complete tier, never a
* currently selected subset. caseMs is the longest single case budget, the
* wall of one isolated case shard. These rows add no case-count or model-work
* policy. The 10-second terms preserve the existing Codex/recording
* finalization grace.
*/
export const FILE_RETRY_BUDGETS = [
...STRICT_RETRY_CASE_BUDGETS,
...[
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000 },
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS },
// Seventeen workflow judges include their 10s recording grace; the other
// seven judges retain 120s. Supervise all 24 and the existing one retry.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 2 },
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
// seven judges retain 120s. Supervise all 24.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 },
// Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s.
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS, configuredRetries: 1 },
].map(({ file, attemptMs, caseMs, configuredRetries }) => {
const retries = retriesWithinCaseCap(caseMs, configuredRetries);
return {
file, attemptMs, caseMs, retries,
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
};
}),
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS },
].map(({ file, attemptMs, caseMs }) => ({
file, attemptMs, caseMs,
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: attemptMs + SHARD_RESERVE_MS,
})),
];
/** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */
+201 -1
View File
@@ -14,8 +14,20 @@ import {
formatComparison,
generateCommentary,
judgePassed,
CONTRACT_VIOLATIONS_FILE,
ContractViolation,
TRIAL_ENV,
TRIAL_OUTCOME_SCHEMA,
expectContract,
failureClassOf,
formatTrialOutcomes,
panelVerdict,
parseTrialOutcomes,
sanitizeTrialError,
trialContextFromEnv,
} from './eval-store';
import type { EvalResult, EvalTestEntry, ComparisonResult } from './eval-store';
import type { EvalResult, EvalTestEntry, ComparisonResult, PanelTrial, TrialOutcomeRecord } from './eval-store';
import { EVAL_POLICY } from './periodic-exclude-data';
import { manualReviewFixture } from './manual-judge-review-fixture';
let tmpDir: string;
@@ -957,3 +969,191 @@ describe('generateCommentary', () => {
expect(notes.some(n => n.includes('Stable run'))).toBe(true);
});
});
// --- Trials, panel verdicts and contract vetoes (eval reliability policy) ---
const PANEL = EVAL_POLICY.panel;
const pass = (trial: number, extra: Partial<PanelTrial> = {}): PanelTrial => ({ trial, outcome: 'passed', ...extra });
const fail = (trial: number, extra: Partial<PanelTrial> = {}): PanelTrial => ({ trial, outcome: 'failed', ...extra });
const behavior = (trials: PanelTrial[], quarantined = false) =>
panelVerdict({ case: 'case-x', kind: 'behavior', panel: PANEL, trials, quarantined });
describe('panelVerdict', () => {
test('policy constants are the approved pre-registration', () => {
expect(EVAL_POLICY.panel).toEqual({ n: 3, k: 2 });
expect(EVAL_POLICY.quarantine).toEqual({
entry: { rate: 0.95, minTrials: 10 },
exit: { rate: 0.97, minTrials: 10 },
capFraction: 0.1,
expiryWeeklyRuns: 8,
});
expect(EVAL_POLICY.infraRedispatch).toBe(1);
});
test('rule: one trial, any failure fails the lane', () => {
const ok = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 1, k: 1 }, trials: [pass(1)] });
expect(ok).toMatchObject({ status: 'PASS', split: false, failsLane: false, coverage: true, marks: '✓' });
const bad = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 1, k: 1 }, trials: [fail(1)] });
expect(bad).toMatchObject({ status: 'FAIL', failsLane: true, coverage: false, redClass: 'VERDICT', marks: '✗' });
});
test('behavior 3/3 is a clean PASS', () => {
expect(behavior([pass(1), pass(2), pass(3)])).toMatchObject({ status: 'PASS', split: false, passed: 3, failsLane: false });
});
test('behavior 2/3 is a split PASS that shows its failed trial', () => {
const v = behavior([pass(1), fail(2, { exit_reason: 'timeout' }), pass(3)]);
expect(v).toMatchObject({ status: 'PASS', split: true, passed: 2, failed: 1, failsLane: false, coverage: true, marks: '✓✗✓', reason: 'PASS 2/3' });
expect(v.trials[1].exit_reason).toBe('timeout');
});
test('behavior 1/3 and 0/3 fail the lane', () => {
expect(behavior([pass(1), fail(2), fail(3)])).toMatchObject({ status: 'FAIL', failsLane: true, redClass: 'VERDICT' });
expect(behavior([fail(1), fail(2), fail(3)])).toMatchObject({ status: 'FAIL', failsLane: true });
});
test('a contract trial fails the panel even at 2/3', () => {
const v = behavior([pass(1), pass(2), fail(3, { failure_class: 'contract' })]);
expect(v).toMatchObject({ status: 'FAIL', contract: true, failsLane: true, redClass: 'VERDICT', reason: 'contract violation' });
});
test('a missing trial is INCOMPLETE and fails the lane', () => {
const v = behavior([pass(1), pass(3)]);
expect(v).toMatchObject({ status: 'INCOMPLETE', failsLane: true, coverage: false, redClass: 'INCOMPLETE', marks: '✓·✓' });
expect(v.reason).toContain('missing trial t2');
});
test('duplicate or out-of-range trial records are INCOMPLETE, never deduplicated', () => {
expect(behavior([pass(1), pass(2), pass(2), fail(3)]).status).toBe('INCOMPLETE');
expect(behavior([pass(1), pass(2), pass(3), pass(4)]).reason).toContain('unexpected trial t4');
const v = panelVerdict({ case: 'c', kind: 'behavior', panel: { n: 11, k: 6 }, trials: Array.from({ length: 10 }, (_, i) => pass(i + 2)) });
expect(v.reason).toContain('missing trial t1');
});
test('timeout and infra trials count as failed, never passing', () => {
const v = behavior([pass(1), fail(2, { exit_reason: 'timeout' }), fail(3, { failure_class: 'infra' })]);
expect(v).toMatchObject({ status: 'FAIL', passed: 1, failed: 2, failsLane: true, redClass: 'VERDICT' });
const infra = behavior([pass(1), fail(2, { failure_class: 'infra' }), fail(3, { failure_class: 'infra' })]);
expect(infra).toMatchObject({ status: 'FAIL', redClass: 'INFRA' });
expect(failureClassOf({ exit_reason: 'timeout' })).toBe('timeout');
expect(failureClassOf({})).toBe('assertion');
});
test('quarantined: 1/3 does not fail the lane, 0/3 and contract do, no coverage credit', () => {
expect(behavior([pass(1), fail(2), fail(3)], true)).toMatchObject({ status: 'FAIL', failsLane: false, coverage: false, redClass: null });
expect(behavior([fail(1), fail(2), fail(3)], true)).toMatchObject({ status: 'FAIL', failsLane: true });
expect(behavior([pass(1), pass(2), fail(3, { failure_class: 'contract' })], true)).toMatchObject({ status: 'FAIL', failsLane: true });
expect(behavior([pass(1), pass(2), pass(3)], true)).toMatchObject({ status: 'PASS', coverage: false, failsLane: false });
expect(behavior([pass(1), pass(2)], true)).toMatchObject({ status: 'INCOMPLETE', failsLane: true });
});
test('quarantined rule keeps rule meaning (k = n)', () => {
const v = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 3, k: 3 }, trials: [pass(1), pass(2), fail(3)], quarantined: true });
expect(v).toMatchObject({ status: 'FAIL', failsLane: false });
});
test('all-skipped panel is SKIPPED with no credit; partly skipped is INCOMPLETE', () => {
const skip = (trial: number): PanelTrial => ({ trial, outcome: 'skipped' });
expect(behavior([skip(1), skip(2), skip(3)])).toMatchObject({ status: 'SKIPPED', coverage: false, failsLane: false });
expect(behavior([pass(1), pass(2), skip(3)])).toMatchObject({ status: 'INCOMPLETE', failsLane: true });
});
test('trials of different run attempts are never merged into one verdict', () => {
expect(() => behavior([pass(1), pass(2), fail(3, { attempt: 2 })])).toThrow(/run attempts/);
expect(behavior([pass(1, { attempt: 2 }), pass(2, { attempt: 2 }), pass(3, { attempt: 2 })]).attempt).toBe(2);
});
test('invalid panels throw', () => {
expect(() => panelVerdict({ case: 'c', kind: 'behavior', panel: { n: 3, k: 4 }, trials: [] })).toThrow(/invalid panel/);
expect(() => panelVerdict({ case: 'c', kind: 'nope' as any, panel: { n: 1, k: 1 }, trials: [] })).toThrow(/unknown kind/);
});
});
describe('trial context and expectContract', () => {
let dir: string;
const saved: Record<string, string | undefined> = {};
const keys = [...Object.values(TRIAL_ENV), 'GSTACK_EVAL_DIR'];
beforeEach(() => {
dir = fs.mkdtempSync(path.join(os.tmpdir(), 'panel-verdict-'));
for (const key of keys) saved[key] = process.env[key];
});
afterEach(() => {
for (const key of keys) {
if (saved[key] === undefined) delete process.env[key];
else process.env[key] = saved[key];
}
fs.rmSync(dir, { recursive: true, force: true });
});
const setTrial = () => Object.assign(process.env, {
[TRIAL_ENV.caseId]: 'case-x', [TRIAL_ENV.kind]: 'behavior', [TRIAL_ENV.trial]: '2',
[TRIAL_ENV.panelN]: '3', [TRIAL_ENV.panelK]: '2', [TRIAL_ENV.policyVersion]: String(EVAL_POLICY.version),
GSTACK_EVAL_DIR: dir,
});
test('trialContextFromEnv: absent, complete, and malformed', () => {
for (const key of Object.values(TRIAL_ENV)) delete process.env[key];
expect(trialContextFromEnv()).toBeNull();
setTrial();
expect(trialContextFromEnv()).toEqual({ case_id: 'case-x', kind: 'behavior', trial: 2, panel: { n: 3, k: 2 }, policy_version: EVAL_POLICY.version });
process.env[TRIAL_ENV.trial] = '4';
expect(() => trialContextFromEnv()).toThrow(/Malformed trial context/);
});
test('passing contract is a no-op', () => {
setTrial();
expect(() => expectContract(true, 'fine')).not.toThrow();
expect(fs.existsSync(path.join(dir, CONTRACT_VIOLATIONS_FILE))).toBe(false);
});
test('failed contract stamps the recorded entry and the sidecar before throwing', () => {
setTrial();
const collector = new EvalCollector('e2e', dir);
collector.addTest({ name: 'case-x', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 });
expect(() => expectContract(false, 'handoff missing', { collector, name: 'case-x' })).toThrow(ContractViolation);
const partial = JSON.parse(fs.readFileSync(path.join(dir, '_partial-e2e.json'), 'utf-8'));
expect(partial.tests[0]).toMatchObject({ passed: false, failure_class: 'contract', case_id: 'case-x', trial: 2, kind: 'behavior', panel: { n: 3, k: 2 } });
const sidecar = fs.readFileSync(path.join(dir, CONTRACT_VIOLATIONS_FILE), 'utf-8').trim().split('\n').map((l) => JSON.parse(l));
expect(sidecar).toEqual([expect.objectContaining({ case_id: 'case-x', trial: 2, message: 'handoff missing' })]);
});
test('a contract marked before recording stamps the later record, or becomes its own at finalize', async () => {
setTrial();
const collector = new EvalCollector('e2e', dir);
expect(() => expectContract(0, 'no question asked', { collector, name: 'later' })).toThrow('CONTRACT: no question asked');
collector.addTest({ name: 'later', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 });
expect(() => expectContract(null, 'never recorded', { collector, name: 'orphan' })).toThrow();
const file = await collector.finalize();
const tests = JSON.parse(fs.readFileSync(file, 'utf-8')).tests;
expect(tests.find((t: any) => t.name === 'later')).toMatchObject({ passed: false, failure_class: 'contract' });
expect(tests.find((t: any) => t.name === 'orphan')).toMatchObject({ passed: false, failure_class: 'contract', error: 'never recorded' });
});
});
describe('trial-outcomes JSONL', () => {
const record = (extra: Partial<TrialOutcomeRecord> = {}): TrialOutcomeRecord => ({
schema: TRIAL_OUTCOME_SCHEMA, case: 'case-x', file: 'test/x.test.ts', tier: 'gate', kind: 'behavior',
trial: 1, panel: { n: 3, k: 2 }, attempt: 1, outcome: 'passed', duration_ms: 10, cost_usd: 0.1,
policy_version: EVAL_POLICY.version, quarantined: false, execution: 'executed', source: 'shard', ...extra,
});
test('round-trips valid records', () => {
const records = [record(), record({ trial: 2, outcome: 'failed', failure_class: 'timeout', exit_reason: 'timeout' })];
expect(parseTrialOutcomes(formatTrialOutcomes(records))).toEqual({ records, errors: [] });
});
test('writer fails closed; reader reports bad lines as data errors', () => {
expect(() => formatTrialOutcomes([record({ outcome: 'failed' })])).toThrow(/failed without failure_class/);
expect(() => formatTrialOutcomes([record({ trial: 4 })])).toThrow(/trial invalid/);
const text = `${JSON.stringify(record())}\nnot json\n${JSON.stringify({ ...record(), schema: 'other' })}\n`;
const parsed = parseTrialOutcomes(text);
expect(parsed.records).toHaveLength(1);
expect(parsed.errors).toEqual(['line 2: not JSON', 'line 3: schema other']);
expect(parseTrialOutcomes(text, { maxBytes: 10 }).errors[0]).toContain('exceed');
});
test('sanitizeTrialError keeps one capped line without mentions', () => {
expect(sanitizeTrialError('\n expected @garrytan to `see`\nsecond')).toBe("expected @\u200bgarrytan to 'see'");
expect(sanitizeTrialError('x'.repeat(1000))!.length).toBe(300);
expect(sanitizeTrialError('')).toBeUndefined();
});
});
+367 -1
View File
@@ -76,6 +76,18 @@ export interface EvalTestEntry {
* its body again and re-records under the same name. Set by addTest. */
attempt?: number;
// Trial identity (eval reliability policy). Stamped by addTest from the
// TRIAL_ENV variables the paid runner sets on an isolated trial shard.
/** Registry id (E2E_TIERS / LLM_JUDGE_TOUCHFILES key) this record belongs to. */
case_id?: string;
kind?: EvalCaseKind;
/** 1-based trial index within the case's panel. */
trial?: number;
panel?: PanelShape;
/** Why a failed record failed; 'contract' comes only from expectContract. */
failure_class?: TrialFailureClass;
policy_version?: number;
// E2E
transcript?: any[];
prompt?: string;
@@ -131,6 +143,329 @@ export function evalEntryOutcome(entry: unknown): 'passed' | 'failed' | 'manual-
return result.passed === true ? 'passed' : 'failed';
}
// --- Trials and panel verdicts ---
//
// Paid evals never retry. Each case's kind (E2E_KINDS) fixes its trials before
// the run; a panel verdict is computed once, by panelVerdict(), from exactly
// panel.n trial records of one run attempt. The report, collector-outcomes,
// the PR comment and pass-rates all read that one function.
export type EvalCaseKind = 'rule' | 'behavior' | 'judge';
/** assertion: an ordinary failed expectation. contract: expectContract() fired
* (fails the panel at any count). timeout: the case budget ran out.
* infra: API/CLI/runner failure before the model could be graded. */
export type TrialFailureClass = 'assertion' | 'contract' | 'timeout' | 'infra';
export type TrialOutcome = 'passed' | 'failed' | 'skipped';
export interface PanelShape { n: number; k: number }
/** Environment the paid runner sets on an isolated trial shard. */
export const TRIAL_ENV = {
caseId: 'GSTACK_EVAL_CASE_ID',
kind: 'GSTACK_EVAL_KIND',
trial: 'GSTACK_EVAL_TRIAL',
panelN: 'GSTACK_EVAL_PANEL_N',
panelK: 'GSTACK_EVAL_PANEL_K',
policyVersion: 'GSTACK_EVAL_POLICY_VERSION',
} as const;
/** Sidecar every expectContract() failure appends to (in GSTACK_EVAL_DIR), so a
* contract veto survives a test that throws before recording its entry. */
export const CONTRACT_VIOLATIONS_FILE = 'contract-violations.jsonl';
export interface TrialContext {
case_id: string;
kind: EvalCaseKind;
trial: number;
panel: PanelShape;
policy_version: number;
}
const EVAL_KINDS: readonly EvalCaseKind[] = ['rule', 'behavior', 'judge'];
const FAILURE_CLASSES: readonly TrialFailureClass[] = ['assertion', 'contract', 'timeout', 'infra'];
function positiveInt(raw: string | undefined): number | null {
if (raw === undefined || !/^[1-9][0-9]*$/.test(raw)) return null;
return Number(raw);
}
/** Trial context of this process, or null outside an isolated trial shard.
* A partial or malformed context throws: a mislabeled trial is fail-open. */
export function trialContextFromEnv(env: NodeJS.ProcessEnv = process.env): TrialContext | null {
const caseId = env[TRIAL_ENV.caseId];
if (!caseId) return null;
const kind = env[TRIAL_ENV.kind] as EvalCaseKind | undefined;
const trial = positiveInt(env[TRIAL_ENV.trial]);
const n = positiveInt(env[TRIAL_ENV.panelN]);
const k = positiveInt(env[TRIAL_ENV.panelK]);
const policy = positiveInt(env[TRIAL_ENV.policyVersion]);
if (!kind || !EVAL_KINDS.includes(kind) || trial === null || n === null || k === null || policy === null || k > n || trial > n) {
throw new Error(`Malformed trial context for ${caseId}: ${Object.values(TRIAL_ENV).map((name) => `${name}=${env[name] ?? ''}`).join(' ')}`);
}
return { case_id: caseId, kind, trial, panel: { n, k }, policy_version: policy };
}
/** Failure class of a failed record: an explicit class wins, then the exit reason. */
export function failureClassOf(entry: Pick<EvalTestEntry, 'failure_class' | 'exit_reason'>): TrialFailureClass {
if (entry.failure_class && FAILURE_CLASSES.includes(entry.failure_class)) return entry.failure_class;
return entry.exit_reason === 'timeout' ? 'timeout' : 'assertion';
}
export class ContractViolation extends Error {
constructor(message: string) {
super(`CONTRACT: ${message}`);
this.name = 'ContractViolation';
}
}
/**
* Assert a contract: an outcome the product must meet on every run. On failure
* it records failure_class 'contract' before throwing, both on the collector
* entry named `record.name` (now or when the test records it) and in the
* GSTACK_EVAL_DIR sidecar, so panelVerdict() fails the panel even at 2 of 3.
*/
export function expectContract(
condition: unknown,
message: string,
record?: { collector: EvalCollector | null; name: string },
): asserts condition {
if (condition) return;
record?.collector?.markContractViolation(record.name, message);
const evalDir = process.env.GSTACK_EVAL_DIR;
if (evalDir) {
const context = trialContextFromEnv();
fs.mkdirSync(evalDir, { recursive: true });
fs.appendFileSync(path.join(evalDir, CONTRACT_VIOLATIONS_FILE), JSON.stringify({
case_id: context?.case_id ?? record?.name ?? null,
name: record?.name ?? null,
trial: context?.trial ?? null,
message,
at: new Date().toISOString(),
}) + '\n');
}
throw new ContractViolation(message);
}
export interface PanelTrial {
trial: number;
outcome: TrialOutcome;
/** Required meaning for a failed trial; absent reads as 'assertion'. */
failure_class?: TrialFailureClass;
/** CI run attempt (github.run_attempt); absent means 1. */
attempt?: number;
exit_reason?: string;
error?: string;
execution?: 'executed' | 'reused';
}
export interface PanelVerdictInput {
case: string;
kind: EvalCaseKind;
panel: PanelShape;
trials: readonly PanelTrial[];
quarantined?: boolean;
}
export type PanelStatus = 'PASS' | 'FAIL' | 'INCOMPLETE' | 'SKIPPED';
export interface PanelVerdict {
case: string;
kind: EvalCaseKind;
panel: PanelShape;
attempt: number;
quarantined: boolean;
status: PanelStatus;
passed: number;
failed: number;
/** A failed trial carried failure_class 'contract'. */
contract: boolean;
/** PASS with at least one failed trial: shown as `PASS k/n`, never clean. */
split: boolean;
/** Whether this verdict makes the lane red. */
failsLane: boolean;
/** Whether it counts as passing coverage (never for quarantined or skipped). */
coverage: boolean;
/** Machine classification of a lane-failing verdict: INCOMPLETE (missing or
* malformed trial records), INFRA (every failed trial is infra-class), or
* VERDICT (a real red). Null when the verdict does not fail the lane. */
redClass: 'INCOMPLETE' | 'INFRA' | 'VERDICT' | null;
/** One glyph per trial index: ✓ pass, ✗ fail, – skipped, · missing. */
marks: string;
reason: string;
trials: PanelTrial[];
}
/**
* The single verdict function. `rule`/`judge` cases run panel {1,1}; `behavior`
* cases run EVAL_POLICY.panel; a quarantined case runs a full panel whose k
* keeps its kind's meaning (k = n for rule). Verdict: INCOMPLETE unless
* exactly one record per trial index 1..n; SKIPPED when every trial skipped;
* FAIL on any contract trial; otherwise PASS iff passed >= k. A quarantined
* FAIL fails the lane only on a hard break (0 of n) or a contract violation.
*/
export function panelVerdict(input: PanelVerdictInput): PanelVerdict {
const { n, k } = input.panel;
if (!Number.isInteger(n) || !Number.isInteger(k) || n < 1 || k < 1 || k > n) {
throw new Error(`${input.case}: invalid panel {n:${n}, k:${k}}`);
}
if (!EVAL_KINDS.includes(input.kind)) throw new Error(`${input.case}: unknown kind ${String(input.kind)}`);
const attempts = new Set(input.trials.map((t) => t.attempt ?? 1));
if (attempts.size > 1) {
throw new Error(`${input.case}: trials from run attempts ${[...attempts].join(', ')}; compute one verdict per attempt`);
}
const attempt = [...attempts][0] ?? 1;
const quarantined = input.quarantined === true;
const trials = [...input.trials].sort((a, b) => a.trial - b.trial);
const byIndex = new Map<number, PanelTrial>();
const problems: string[] = [];
for (const t of trials) {
if (!Number.isInteger(t.trial) || t.trial < 1 || t.trial > n) problems.push(`unexpected trial t${t.trial}`);
else if (byIndex.has(t.trial)) problems.push(`duplicate trial t${t.trial}`);
else if (t.outcome !== 'passed' && t.outcome !== 'failed' && t.outcome !== 'skipped') problems.push(`t${t.trial} has outcome ${String(t.outcome)}`);
else byIndex.set(t.trial, t);
}
for (let i = 1; i <= n; i++) if (!trials.some((t) => t.trial === i)) problems.push(`missing trial t${i}`);
const marks = Array.from({ length: n }, (_, i) => {
const t = byIndex.get(i + 1);
return !t ? '·' : t.outcome === 'passed' ? '✓' : t.outcome === 'failed' ? '✗' : '–';
}).join('');
const passed = [...byIndex.values()].filter((t) => t.outcome === 'passed').length;
const failedTrials = [...byIndex.values()].filter((t) => t.outcome === 'failed');
const skipped = [...byIndex.values()].filter((t) => t.outcome === 'skipped').length;
const contract = failedTrials.some((t) => failureClassOf(t) === 'contract');
const base = { case: input.case, kind: input.kind, panel: { n, k }, attempt, quarantined, passed, failed: failedTrials.length, contract, marks, trials };
if (problems.length === 0 && skipped === n) {
return { ...base, status: 'SKIPPED', split: false, failsLane: false, coverage: false, redClass: null, reason: 'every trial skipped (no verdict credit)' };
}
if (problems.length === 0 && skipped > 0) problems.push(`${skipped} of ${n} trials skipped`);
if (problems.length > 0) {
return { ...base, status: 'INCOMPLETE', split: false, failsLane: true, coverage: false, redClass: 'INCOMPLETE', reason: problems.join('; ') };
}
if (!contract && passed >= k) {
const split = failedTrials.length > 0;
return {
...base, status: 'PASS', split, failsLane: false, coverage: !quarantined, redClass: null,
reason: split ? `PASS ${passed}/${n}` : `${passed}/${n} passed`,
};
}
const hardBreak = passed === 0;
const failsLane = !quarantined || contract || hardBreak;
const allInfra = !contract && failedTrials.length > 0 && failedTrials.every((t) => failureClassOf(t) === 'infra');
const why = contract ? 'contract violation' : `${passed}/${n} passed, needs ${k}`;
return {
...base, status: 'FAIL', split: false, failsLane, coverage: false,
redClass: failsLane ? (allInfra ? 'INFRA' : 'VERDICT') : null,
reason: !quarantined ? why
: contract ? `${why}; quarantine never excuses a contract`
: hardBreak ? `${why}; quarantined hard break`
: `${why}; quarantined, does not fail the lane`,
};
}
// --- trial-outcomes JSONL (one line per trial; pass-rate history input) ---
export const TRIAL_OUTCOME_SCHEMA = 'gstack-trial-outcome/v1';
export const TRIAL_OUTCOMES_FILE = 'trial-outcomes.jsonl';
/** Cap on a stored `error` line (sanitized first line of the failure). */
export const TRIAL_ERROR_MAX = 300;
export interface TrialOutcomeRecord {
schema: typeof TRIAL_OUTCOME_SCHEMA;
/** Registry id. */
case: string;
file: string;
tier: string;
kind: EvalCaseKind;
trial: number;
panel: PanelShape;
/** CI run attempt (github.run_attempt); 1 locally and for pre-policy backfill. */
attempt: number;
outcome: TrialOutcome;
/** Present exactly when outcome is 'failed'. */
failure_class?: TrialFailureClass;
exit_reason?: string;
error?: string;
duration_ms: number;
cost_usd: number;
model?: string;
cli_version?: string;
/** Reuse input key of the trial's shard, when known. */
input_identity?: string;
/** EVAL_POLICY.version; 0 marks pre-policy backfill. */
policy_version: number;
quarantined: boolean;
execution: 'executed' | 'reused';
/** shard: isolated trial shard status. junit: a rule file shard's per-test
* JUnit outcome. backfill: imported pre-policy artifact record. */
source: 'shard' | 'junit' | 'backfill';
run_id?: string;
sha?: string;
lane?: string;
recorded_at?: string;
/** History series key: a hash of the case's own touchfiles (GLOBAL_TOUCHFILES excluded), stamped by the report job. */
series_identity?: string;
}
/** First line of free text, stripped of @-mentions and control characters, capped. */
export function sanitizeTrialError(text: string | undefined): string | undefined {
if (!text) return undefined;
const first = text.split('\n').map((l) => l.trim()).find((l) => l.length > 0);
if (!first) return undefined;
// eslint-disable-next-line no-control-regex
const clean = first.replace(/[\u0000-\u001f\u007f]/g, ' ').replace(/`/g, "'").replace(/@(?=[A-Za-z0-9_-])/g, '@\u200b');
return clean.length > TRIAL_ERROR_MAX ? `${clean.slice(0, TRIAL_ERROR_MAX - 1)}…` : clean;
}
function trialRecordProblems(r: any): string[] {
const problems: string[] = [];
if (!r || typeof r !== 'object' || Array.isArray(r)) return ['not an object'];
if (r.schema !== TRIAL_OUTCOME_SCHEMA) problems.push(`schema ${String(r.schema)}`);
for (const key of ['case', 'file', 'tier'] as const) if (typeof r[key] !== 'string' || r[key].length === 0) problems.push(`${key} missing`);
if (!EVAL_KINDS.includes(r.kind)) problems.push(`kind ${String(r.kind)}`);
const n = r.panel?.n, k = r.panel?.k;
if (!Number.isInteger(n) || !Number.isInteger(k) || n < 1 || k < 1 || k > n) problems.push('panel invalid');
if (!Number.isInteger(r.trial) || r.trial < 1 || (Number.isInteger(n) && r.trial > n)) problems.push('trial invalid');
if (!Number.isInteger(r.attempt) || r.attempt < 1) problems.push('attempt invalid');
if (!['passed', 'failed', 'skipped'].includes(r.outcome)) problems.push(`outcome ${String(r.outcome)}`);
if (r.outcome === 'failed' && !FAILURE_CLASSES.includes(r.failure_class)) problems.push('failed without failure_class');
if (r.outcome !== 'failed' && r.failure_class !== undefined) problems.push('failure_class on a non-failed trial');
if (typeof r.duration_ms !== 'number' || !Number.isFinite(r.duration_ms) || r.duration_ms < 0) problems.push('duration_ms invalid');
if (typeof r.cost_usd !== 'number' || !Number.isFinite(r.cost_usd) || r.cost_usd < 0) problems.push('cost_usd invalid');
if (!Number.isInteger(r.policy_version) || r.policy_version < 0) problems.push('policy_version invalid');
if (typeof r.quarantined !== 'boolean') problems.push('quarantined invalid');
if (r.execution !== 'executed' && r.execution !== 'reused') problems.push('execution invalid');
if (!['shard', 'junit', 'backfill'].includes(r.source)) problems.push('source invalid');
if (r.error !== undefined && (typeof r.error !== 'string' || r.error.length > TRIAL_ERROR_MAX)) problems.push('error invalid');
if (r.series_identity !== undefined && (typeof r.series_identity !== 'string' || !/^[\w.-]{1,64}$/.test(r.series_identity))) problems.push('series_identity invalid');
return problems;
}
/** Serialize records as JSONL; throws on any invalid record (writers fail closed). */
export function formatTrialOutcomes(records: readonly TrialOutcomeRecord[]): string {
return records.map((r) => {
const problems = trialRecordProblems(r);
if (problems.length > 0) throw new Error(`invalid trial record ${r?.case}~t${r?.trial}: ${problems.join(', ')}`);
return JSON.stringify(r);
}).join('\n') + (records.length > 0 ? '\n' : '');
}
/** Parse downloaded JSONL as data only: invalid lines are reported, never guessed. */
export function parseTrialOutcomes(text: string, opts: { maxBytes?: number } = {}): { records: TrialOutcomeRecord[]; errors: string[] } {
const maxBytes = opts.maxBytes ?? 16 * 1024 * 1024;
if (Buffer.byteLength(text) > maxBytes) return { records: [], errors: [`trial outcomes exceed ${maxBytes} bytes`] };
const records: TrialOutcomeRecord[] = [];
const errors: string[] = [];
text.split('\n').forEach((line, i) => {
if (line.trim() === '') return;
let parsed: unknown;
try { parsed = JSON.parse(line); } catch { errors.push(`line ${i + 1}: not JSON`); return; }
const problems = trialRecordProblems(parsed);
if (problems.length > 0) errors.push(`line ${i + 1}: ${problems.join(', ')}`);
else records.push(parsed as TrialOutcomeRecord);
});
return { records, errors };
}
export interface EvalResult {
schema_version: number;
version: string;
@@ -887,6 +1222,7 @@ export class EvalCollector {
private shard: string | null;
private fileNamespace?: string;
private createdAt = Date.now();
private pendingContract = new Map<string, string>();
constructor(tier: 'e2e' | 'llm-judge', evalDir?: string, fileNamespace?: string) {
if (fileNamespace !== undefined && !/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(fileNamespace)) {
@@ -903,7 +1239,29 @@ export class EvalCollector {
// names are unique by convention). Stamp the 1-based attempt so a
// pass-on-attempt-2 stays visible forever — the stream hides it.
const prior = this.tests.filter((t) => t.name === entry.name).length;
this.tests.push({ ...entry, attempt: prior + 1 });
const context = trialContextFromEnv();
const record: EvalTestEntry = { ...(context ?? {}), ...entry, attempt: prior + 1 };
const contract = this.pendingContract.get(entry.name);
if (contract !== undefined) {
this.pendingContract.delete(entry.name);
Object.assign(record, { passed: false, failure_class: 'contract', error: record.error ?? contract });
}
this.tests.push(record);
this.savePartial();
}
/** expectContract() hook: mark `name`'s latest record (or its next one) as a
* contract failure. An unmatched mark becomes its own failed record at
* finalize, so the veto is never lost. */
markContractViolation(name: string, message: string): void {
const existing = this.tests.filter((t) => t.name === name).at(-1);
if (!existing) {
this.pendingContract.set(name, message);
return;
}
existing.passed = false;
existing.failure_class = 'contract';
existing.error = existing.error ?? message;
this.savePartial();
}
@@ -959,6 +1317,14 @@ export class EvalCollector {
async finalize(): Promise<string> {
if (this.finalized) return '';
this.finalized = true;
for (const [name, message] of this.pendingContract) {
this.tests.push({
...(trialContextFromEnv() ?? {}),
name, suite: 'contract', tier: this.tier, passed: false, duration_ms: 0, cost_usd: 0,
failure_class: 'contract', error: message, attempt: 1,
});
}
this.pendingContract.clear();
const git = getGitInfo();
const version = getVersion();
+59
View File
@@ -23,6 +23,8 @@ export interface JudgeScore {
reasoning: string;
}
export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const;
export interface JudgeRefusalEvidence {
stop_reason: 'refusal';
response_id: string | null;
@@ -196,6 +198,63 @@ export async function callJudge<T>(
}
}
/**
* Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this
* helper (imported by many paid tests) does not pull the quarantine registry
* into their touchfile closure. test/judge-panel.test.ts pins the two equal.
*/
export const JUDGE_PANEL_SAMPLES = 3;
/**
* Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of
* independent samples of the SAME prompt concurrently, inside its unchanged
* JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
* against the unchanged minimum; boolean fields gate on a strict majority.
* A sample that errors (refusal, truncation, non-JSON, malformed field) fails
* the whole panel and is never resampled. callJudge's 429 backoff happens
* before any model output exists, so it is transport, not a verdict retry.
*/
export async function judgePanel<T>(sample: () => Promise<T>): Promise<T[]> {
const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample()));
const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []);
if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult<T>).value);
const first = failures[0]!;
// A refusal is an unscored panel only when EVERY sample refused; a partial
// refusal beside scored samples is an ordinary failed panel.
if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) {
throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`);
}
throw first.reason;
}
/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */
export function judgePanelMean<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, keys: readonly K[]): Record<K, number> {
if (samples.length === 0) throw new Error('Judge panel has no samples');
return Object.fromEntries(keys.map(key => {
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value));
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`);
return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length];
})) as Record<K, number>;
}
/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */
export function judgePanelMajority<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, key: K): boolean {
if (samples.length === 0) throw new Error('Judge panel has no samples');
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
const bad = values.findIndex(value => typeof value !== 'boolean');
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`);
return values.filter(value => value === true).length * 2 > values.length;
}
/** Sample reasoning lines, numbered, for the collector record. */
export function judgePanelReasoning(samples: ReadonlyArray<unknown>): string {
return samples.map((sample, index) => {
const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined;
return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`;
}).join('\n');
}
/**
* Score documentation quality on clarity/completeness/actionability (1-5).
*/
+64
View File
@@ -59,3 +59,67 @@ export const CASE_CI_EXCLUDE: Record<string, { reason: string; tracking: string
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
};
/**
* Paid-eval verdict policy, pre-registered (approved 2026-09-29). Frozen before
* the census: any change after seeing census results needs Garry's
* re-approval and a fresh census, and bumps `version` (every trial record
* carries it as policy_version, so pass-rate history segments at the change).
* panel - behavior cases and quarantined cases run n independent
* trials; a behavior panel PASSES at >= k passing trials with
* no contract violation. Rule and judge cases run one trial.
* quarantine - entry below `entry.rate` per trial over >= `entry.minTrials`
* new-policy trials; exit at >= `exit.rate` over >=
* `exit.minTrials`; at most `capFraction` of each tier's
* blocking cases; an entry expires after `expiryWeeklyRuns`.
* judge - a judge case draws `samples` independent samples of one
* prompt concurrently; numeric dimensions gate on the panel
* mean against the unchanged threshold, booleans on a strict
* majority; an erroring sample fails the panel, never resampled.
* drift - one-sided Fisher exact alarm between input-identity series
* (Holm-controlled across the cases tested in one report).
* infraRedispatch - a census whose every red verdict is machine-classified
* INFRA or INCOMPLETE may be re-dispatched this many times as
* a new run; both runs are reported.
*/
export const EVAL_POLICY = {
version: 1,
panel: { n: 3, k: 2 },
quarantine: {
entry: { rate: 0.95, minTrials: 10 },
exit: { rate: 0.97, minTrials: 10 },
capFraction: 0.10,
expiryWeeklyRuns: 8,
},
judge: { samples: 3 },
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
infraRedispatch: 1,
} as const;
/**
* Quarantined paid cases, keyed by registry id (an E2E_TIERS key). A
* quarantined case still runs its full panel and reports in every lane, but
* its failed verdict cannot fail the lane unless the panel is a hard break
* (0 of n) or a trial violated a contract; it never counts as passing
* coverage. An entry needs the entry rule met on the current input identity,
* a written diagnosis that the failures are detector, harness or model-latency
* failures (a product defect is never quarantined), and unchanged case
* touchfiles in the change that adds it. Pinned by
* test/periodic-exclude-policy.test.ts.
* reason - the written diagnosis, with the pass-rate evidence
* failureClass - what the diagnosis found; a product defect has no class here
* tracking - issue or TODOS pointer
* owner - who removes it
* enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here)
* exit - the measurable exit condition
* At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and
* periodic are the blocking tiers) may be quarantined at once.
*/
export const CASE_QUARANTINE: Record<string, {
reason: string;
failureClass: 'detector' | 'harness' | 'model-latency';
tracking: string;
owner: string;
enteredAt: string;
exit: string;
}> = {};
+16 -3
View File
@@ -34,6 +34,8 @@ import {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
} from './touchfiles-data';
/** Repo-relative path of the pure-data file (the map-diff subject). */
@@ -145,6 +147,9 @@ export interface TouchfileMaps {
E2E_TIERS: Record<string, string>;
LLM_JUDGE_TOUCHFILES: Record<string, string[]>;
GLOBAL_TOUCHFILES: string[];
/** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */
E2E_KINDS?: Record<string, string>;
BEHAVIOR_WHY?: Record<string, string>;
}
export type MapDiffCause =
@@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
};
function isStringArray(v: unknown): v is string[] {
@@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps {
return isRecordOfStringArrays(o.E2E_TOUCHFILES)
&& isRecordOfStrings(o.E2E_TIERS)
&& isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES)
&& isStringArray(o.GLOBAL_TOUCHFILES);
&& isStringArray(o.GLOBAL_TOUCHFILES)
&& (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS))
&& (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY));
}
/**
* Pure map-diff core (injectable for tests — no git, no filesystem).
*
* A key counts as CHANGED when it was added to any per-key map, its dep-list
* array differs, or its tier value flipped. A key counts as REMOVED only when
* array differs, or its tier, kind or behavior tolerance changed. A per-key
* map missing on the old side (a base revision older than E2E_KINDS /
* BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when
* it is gone from every new per-key map; a key dropped from one map but still
* present in another (e.g. tier entry deleted, touchfile entry kept) counts
* as changed — conservative, because the test still exists with a different
@@ -212,7 +223,7 @@ export function diffTouchfileMapsCore(
oldMaps: TouchfileMaps,
newMaps: TouchfileMaps,
): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } {
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const;
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const;
const changed = new Set<string>();
const rawRemoved = new Set<string>();
@@ -290,6 +301,8 @@ export function diffTouchfileMaps(
' E2E_TIERS: m.E2E_TIERS,',
' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,',
' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,',
' E2E_KINDS: m.E2E_KINDS,',
' BEHAVIOR_WHY: m.BEHAVIOR_WHY,',
'}));',
'',
].join('\n'));
+303
View File
@@ -1573,3 +1573,306 @@ export const GLOBAL_TOUCHFILES = [
// diffed per key, so a data-only edit runs just the affected tests.
// Map-diff fails CLOSED — any error on that path still runs everything.
];
/**
* Eval kind per live case (every E2E_TIERS and LLM_JUDGE_TOUCHFILES key).
* The kind fixes the trial policy before the run (EVAL_POLICY in
* periodic-exclude-data.ts):
* rule - one trial; any failed assertion fails the verdict. The default.
* behavior - a panel of independent trials, PASS at the policy majority;
* needs a BEHAVIOR_WHY entry naming the tolerated deviation.
* judge - an LLM-judge score of a static input, sampled as a panel.
* Reclassification is a reviewed diff, never a runtime switch.
*/
export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'ship-skipped-queued-finding': 'rule',
'investigate-owned-completion': 'rule',
'investigate-owned-abort': 'rule',
'investigate-owned-ending-error': 'rule',
'shared-libs-review-path-eligibility': 'rule',
'shared-libs-review-index-flags': 'rule',
'shared-libs-review-prior-coverage': 'rule',
'shared-libs-codex-read-only': 'rule',
'shared-libs-read-only': 'rule',
'shared-libs-unsupported-git': 'rule',
'shared-libs-review-lifecycle': 'rule',
'shared-libs-review-revalidation': 'rule',
'shared-libs-opportunity-judgment': 'behavior',
'shared-libs-pr-coverage': 'rule',
'shared-libs-plan-callers': 'rule',
'browse-basic': 'rule',
'browse-snapshot': 'rule',
'aside-browse-basic': 'rule',
'aside-browse-flow': 'rule',
'aside-qa-quick': 'rule',
'aside-scrape-json': 'rule',
'aside-canary-quick': 'rule',
'hermetic-canary': 'rule',
'hermetic-sentinel': 'rule',
'skillmd-setup-discovery': 'rule',
'skillmd-no-local-binary': 'rule',
'skillmd-outside-git': 'rule',
'session-awareness': 'rule',
'operational-learning': 'rule',
'first-task-scaffold': 'rule',
'qa-quick': 'rule',
'qa-b6-static': 'rule',
'qa-b7-spa': 'rule',
'qa-b8-checkout': 'rule',
'qa-only-no-fix': 'rule',
'qa-fix-loop': 'rule',
'qa-bootstrap': 'rule',
'review-exploratory-small-cli': 'rule',
'ship-exploratory-small-cli': 'rule',
'ship-exploratory-unavailable': 'rule',
'ship-exploratory-plan-checks': 'rule',
'ship-exploratory-late-input': 'rule',
'qa-functional-cli-report': 'rule',
'qa-functional-webhook-report': 'rule',
'qa-functional-cli-fix': 'rule',
'qa-functional-webhook-fix': 'rule',
'review-sql-injection': 'rule',
'review-enum-completeness': 'rule',
'review-base-branch': 'rule',
'review-design-lite': 'behavior',
'review-coverage-audit': 'rule',
'review-dashboard-via': 'rule',
'review-army-migration-safety': 'rule',
'review-army-perf-n-plus-one': 'rule',
'review-army-delivery-audit': 'rule',
'review-army-quality-score': 'rule',
'review-army-json-findings': 'rule',
'review-army-red-team': 'behavior',
'review-army-consensus': 'behavior',
'review-army-simplification': 'behavior',
'review-army-simplification-precision': 'behavior',
'office-hours-spec-review': 'rule',
'office-hours-brain-writeback': 'behavior',
'gbrain-roundtrip-local': 'rule',
'sync-gbrain-read-ready': 'rule',
'sync-gbrain-read-unknown': 'rule',
'office-hours-forcing-energy': 'behavior',
'office-hours-builder-wildness': 'behavior',
'plan-ceo-review': 'rule',
'plan-ceo-review-selective': 'rule',
'plan-ceo-review-benefits': 'rule',
'plan-ceo-review-expansion-energy': 'behavior',
'plan-eng-review': 'rule',
'plan-eng-review-artifact': 'rule',
'plan-eng-coverage-audit': 'rule',
'plan-review-report': 'rule',
'plan-ceo-review-plan-mode': 'rule',
'plan-eng-review-plan-mode': 'rule',
'plan-design-review-plan-mode': 'rule',
'plan-devex-review-plan-mode': 'rule',
'plan-mode-no-op': 'rule',
'office-hours-auto-mode': 'rule',
'auto-decide-preserved': 'rule',
'auq-format-gate': 'rule',
'plan-ceo-mode-routing': 'rule',
'plan-design-with-ui-scope': 'rule',
'tpa-present': 'rule',
'tpa-absent-linux': 'rule',
'tpa-broken': 'rule',
'tpa-absent-darwin': 'rule',
'tpa-apple-ban': 'rule',
'ship-section-loading': 'rule',
'plan-ceo-section-loading': 'rule',
'carve-section-loading': 'rule',
'plan-eng-finding-floor': 'rule',
'plan-ceo-finding-floor': 'rule',
'plan-design-finding-floor': 'rule',
'plan-devex-finding-floor': 'rule',
'plan-eng-multi-finding-batching': 'rule',
'plan-ceo-split-overflow': 'rule',
'setup-gbrain-remote': 'rule',
'setup-gbrain-bad-token': 'rule',
'setup-gbrain-path4-local-pglite': 'rule',
'plan-ceo-review-format-mode': 'behavior',
'plan-ceo-review-format-approach': 'behavior',
'plan-eng-review-format-coverage': 'behavior',
'plan-eng-review-format-kind': 'behavior',
'office-hours-phase4-fork': 'behavior',
'llm-judge-recommendation': 'judge',
'plan-ceo-review-prosons-cadence': 'behavior',
'plan-review-prosons-format': 'behavior',
'plan-review-prosons-hardstop-neg': 'behavior',
'plan-review-prosons-neutral-neg': 'behavior',
'plan-tune-inspect': 'rule',
'codex-offered-office-hours': 'rule',
'codex-offered-ceo-review': 'rule',
'codex-offered-design-review': 'rule',
'codex-offered-eng-review': 'rule',
'timeline-event-flow': 'rule',
'context-recovery-artifacts': 'rule',
'context-save-writes-file': 'rule',
'context-restore-loads-latest': 'rule',
'context-save-routing': 'rule',
'context-save-then-restore-roundtrip': 'rule',
'context-restore-fragment-match': 'rule',
'context-restore-empty-state': 'rule',
'context-restore-list-delegates': 'rule',
'context-restore-legacy-compat': 'rule',
'context-save-list-current-branch': 'rule',
'context-save-list-all-branches': 'rule',
'ship-base-branch': 'rule',
'ship-local-workflow': 'rule',
'ship-managed-hook-refresh': 'rule',
'ship-unmanaged-hook-consent': 'rule',
'ship-local-hook-preservation': 'rule',
'ship-coverage-audit': 'rule',
'ship-triage': 'rule',
'ship-docsync-missing-marker': 'rule',
'ship-docsync-missing-asset': 'rule',
'ship-docsync-launch-failure': 'rule',
'ship-docsync-timeout-unsettled': 'rule',
'ship-docsync-late-result': 'rule',
'ship-docsync-stale-before': 'rule',
'ship-docsync-stale-after': 'rule',
'ship-docsync-recovery': 'rule',
'ship-docsync-completion': 'rule',
'ship-docsync-current': 'rule',
'ship-docsync-failure': 'rule',
'ship-docsync-store': 'rule',
'docsync-spawned': 'rule',
'retro': 'rule',
'retro-base-branch': 'rule',
'cso-full-audit': 'rule',
'cso-diff-mode': 'rule',
'cso-infra-scope': 'rule',
'learnings-show': 'rule',
'document-release': 'rule',
'codex-review': 'rule',
'codex-discover-skill': 'rule',
'codex-review-findings': 'rule',
'outside-voice-codex-to-claude-code': 'rule',
'outside-voice-claude-code-to-codex': 'rule',
'outside-plan-disabled-no-fallback': 'rule',
'codex-sol-scope-termination': 'rule',
'design-consultation-core': 'rule',
'design-consultation-existing': 'rule',
'design-consultation-research': 'rule',
'design-consultation-preview': 'rule',
'plan-design-review-no-ui-scope': 'rule',
'design-review-fix': 'rule',
'design-review-detector-shim': 'rule',
'design-review-detector-shim-dom': 'rule',
'design-review-plugin-handoff': 'rule',
'design-html-slop-gate': 'behavior',
'diagram-triplet': 'rule',
'diagram-authoring-quality': 'rule',
'gstack-upgrade-happy-path': 'rule',
'land-and-deploy-workflow': 'rule',
'land-and-deploy-first-run': 'rule',
'land-and-deploy-review-gate': 'rule',
'canary-workflow': 'rule',
'benchmark-workflow': 'rule',
'setup-deploy-workflow': 'rule',
'autoplan-dual-voice': 'rule',
'benchmark-providers-live': 'rule',
'scrape-match-path': 'behavior',
'scrape-prototype-path': 'behavior',
'skillify-happy-path': 'rule',
'skillify-provenance-refusal': 'rule',
'skillify-approval-reject': 'rule',
'journey-ideation': 'rule',
'journey-plan-eng': 'rule',
'journey-debug': 'rule',
'journey-qa': 'rule',
'journey-code-review': 'rule',
'journey-ship': 'rule',
'journey-docs': 'rule',
'journey-retro': 'rule',
'journey-design-system': 'rule',
'journey-visual-qa': 'rule',
'ios-qa-device': 'rule',
'arm-benchmark-native-overbuild': 'rule',
'arm-benchmark-crud-endpoint': 'rule',
'arm-benchmark-bugfix-decoys': 'rule',
'office-hours-section-loading': 'rule',
'office-hours-design-draft': 'rule',
'plan-decision-classification': 'rule',
'plan-devex-peer-comparison-classification': 'rule',
'health-reporting': 'rule',
'overlay-harness-claude-dedicated-tools-vs-bash': 'rule',
'overlay-harness-opus-4-7-effort-match-trivial': 'rule',
'overlay-harness-opus-4-7-literal-interpretation': 'rule',
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule',
'journey-negatives': 'rule',
'review/SKILL.md workflow': 'judge',
'setup-browser-cookies/SKILL.md workflow': 'judge',
'browse/SKILL.md reference': 'judge',
'setup block': 'judge',
'qa/SKILL.md workflow': 'judge',
'qa/SKILL.md health rubric': 'judge',
'qa/SKILL.md anti-refusal': 'judge',
'cross-skill greptile consistency': 'judge',
'ship/SKILL.md workflow': 'judge',
'document-release/SKILL.md workflow': 'judge',
'plan-ceo-review/SKILL.md modes': 'judge',
'plan-eng-review/SKILL.md sections': 'judge',
'plan-design-review/SKILL.md passes': 'judge',
'design-review/SKILL.md fix loop': 'judge',
'design-consultation/SKILL.md research': 'judge',
'land-and-deploy/SKILL.md workflow': 'judge',
'canary/SKILL.md monitoring loop': 'judge',
'benchmark/SKILL.md perf collection': 'judge',
'setup-deploy/SKILL.md platform setup': 'judge',
'retro/SKILL.md instructions': 'judge',
'qa-only/SKILL.md workflow': 'judge',
'gstack-upgrade/SKILL.md upgrade flow': 'judge',
'sync-gbrain/SKILL.md read-only readiness': 'judge',
'voice directive tone': 'judge',
};
/**
* One-line tolerance for every behavior-kind case: why an occasional
* deviation is acceptable product behavior. Keys equal the behavior ids of
* E2E_KINDS; values are non-empty.
*/
export const BEHAVIOR_WHY: Record<string, string> = {
'shared-libs-opportunity-judgment':
"Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.",
'review-design-lite':
"How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.",
'review-army-red-team':
"Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.",
'review-army-consensus':
"Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.",
'review-army-simplification':
"Flagging the planted unnecessary structure is an advisory-lens quality judgment.",
'review-army-simplification-precision':
"Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.",
'office-hours-forcing-energy':
"The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.",
'office-hours-builder-wildness':
"Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.",
'office-hours-brain-writeback':
"The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.",
'office-hours-phase4-fork':
"Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.",
'plan-ceo-review-expansion-energy':
"Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.",
'plan-ceo-review-format-mode':
"Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.",
'plan-ceo-review-format-approach':
"Approach-menu Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-coverage':
"Coverage-issue Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-kind':
"Kind-note wording is live formatting of one AskUserQuestion.",
'plan-ceo-review-prosons-cadence':
"Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.",
'plan-review-prosons-format':
"The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.",
'plan-review-prosons-hardstop-neg':
"Not using the hard-stop escape on an ordinary decision is live formatting of one question.",
'plan-review-prosons-neutral-neg':
"Avoiding neutral posture and naming a because-reason is live formatting of one question.",
'design-html-slop-gate':
"How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.",
'scrape-match-path':
"The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.",
'scrape-prototype-path':
"The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.",
};
+24 -11
View File
@@ -4,7 +4,7 @@ import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
import { JUDGE_MS } from './eval-budgets';
import type { JudgeScore } from './llm-judge';
import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input';
import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache,
type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache';
@@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres
|| typeof value.reasoning !== 'string'
|| (structuredResponse && (!value.reasoning.trim()
|| value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false;
return (['clarity', 'completeness', 'actionability'] as const).every(key =>
return JUDGE_SCORE_DIMENSIONS.every(key =>
typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5);
}
const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 };
/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */
export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array<JudgeScore & EvalCacheValue> } {
if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples'
|| !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES
|| !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false;
const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS);
return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]);
}
export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null;
lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null;
/** The attempt guard is rechecked after synchronous input/provenance reads. */
publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined;
publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined;
} {
const env = opts.env ?? process.env;
const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined };
const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined };
const pr = Number(env.EVALS_CACHE_PR);
// Runtime ID is the immutable CI image manifest, not a mutable image tag.
// Nonstandard Node/Bun preload code or custom model endpoints need a separate
@@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
prompts: { [opts.testName]: prompt },
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1,
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0,
panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' },
...(opts.stream ? { stream: true } : {}),
...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) },
@@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
return {
lookup() {
const result = lookupEvalInputCache({ ...common, identity: before,
validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) });
validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) });
return result.status === 'reused'
? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null;
? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null;
},
publish(scores, isActive = () => true) {
publish(samples, isActive = () => true) {
// Caller reaches here ONLY after its actual assertions passed. A later
// failed case in the file does not erase this independently completed case.
if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) };
if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
const after = currentIdentity();
const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID;
if (!after || !runId || !isActive()) return;
@@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
cancelled: false, skipped: 0, failed: 0, passed: 1,
cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }],
source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() },
result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning },
result: panel,
} });
// A slow synchronous write can consume the recording allowance. The
// caller withdraws this new receipt if its final deadline check fails.
+1 -1
View File
@@ -27,7 +27,7 @@ function runnerDependencies(root: string, entries: string[]): string[] {
const source = fs.readFileSync(file, 'utf8').replace(/^#![^\n]*(?:\n|$)/, '\n');
let audited = source;
if (relative === 'test/helpers/test-selection.ts') {
if (createHash('sha256').update(source).digest('hex') !== '4d2fbcb6249e8d22453d25bfe9b18ee0f4568bbec071918675c38a455d4e1e08') {
if (createHash('sha256').update(source).digest('hex') !== '052ad5a52472bcb41db04c9f21fe6a819e9547768468f7e5d390e0014b567677') {
throw new Error('Re-audit the historical touchfile map loader before excluding its computed import');
}
audited = source.replace('`const m = await import(${JSON.stringify(dataPath)});`,', "'',");
+6 -6
View File
@@ -21,8 +21,8 @@ const fakeEnv = {
};
describe('overlay file policy', () => {
test('grouped planning isolates every overlay and preserves ordinary retries', () => {
// Two short-case files keep their one retry (timeout-is-a-verdict rule).
test('grouped planning isolates every overlay and never retries ordinary files', () => {
// Paid evals never retry (approved 2026-09-29), short-case files included.
const workflow = 'test/skill-e2e-review.test.ts';
const files = [...overlayFiles, 'test/skill-e2e-triage.test.ts', workflow];
for (const maxFilesPerShard of [2, 3, 10]) {
@@ -31,9 +31,9 @@ describe('overlay file policy', () => {
for (const file of overlayFiles) expect(shards).toContainEqual([file]);
const workflowShard = shards.find(shard => shard.includes(workflow))!;
expect(workflowShard.some(isOverlayTestFile)).toBe(false);
expect(retriesForFiles(workflowShard)).toBe(1);
expect(retriesForFiles(workflowShard)).toBe(0);
const args = buildPaidShardArgs(workflowShard, resolvePaidShardTimeoutMs(workflowShard), 2, retriesForFiles(workflowShard));
expect(args[args.indexOf('--retry') + 1]).toBe('1');
expect(args[args.indexOf('--retry') + 1]).toBe('0');
expect(planPaidShards(files.map(file => file.replaceAll('/', '\\')), { maxFilesPerShard })).toEqual(shards);
}
});
@@ -66,10 +66,10 @@ describe('overlay file policy', () => {
for (const file of [normalFile, 'test/skill-e2e-overlay-harness.test.ts', 'test/model-overlays.test.ts']) {
expect(isOverlayTestFile(file)).toBe(false);
expect(resolvePaidShardTimeoutMs([file])).toBe(DEFAULT_SHARD_TIMEOUT_MS);
// Not overlays; unlisted files run once because their case budget is unknown.
// Not overlays; every paid file runs once.
expect(retriesForFiles([file])).toBe(0);
}
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1);
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(0);
expect(resolvePaidShardTimeoutMs([normalFile], 1234)).toBe(1234);
expect(resolvePaidShardTimeoutMs([overlayFiles[0]], 1_900_000)).toBe(1_900_000);
expect(() => resolvePaidShardTimeoutMs([overlayFiles[0]], 1_800_000)).toThrow('explicit wall');
+3 -2
View File
@@ -241,7 +241,7 @@ describe('PR profile paid-runner integration', () => {
expect(guarded[0].status).toBe('passed-empty');
});
test('report distinguishes retained/deferred coverage and final executed/reused outcomes from attempts', () => {
test('report distinguishes retained/deferred coverage and counts every executed/reused record', () => {
const manifest = ceoManifest();
const lines = formatProfileCoverage(manifest).join('\n');
expect(lines).toContain('profile=pr mode=pr');
@@ -252,6 +252,7 @@ describe('PR profile paid-runner integration', () => {
{ name: 'retry', suite: 'judge', passed: true, execution: 'executed' },
{ name: 'cached', suite: 'judge', passed: true, execution: 'reused' },
{ name: 'failed', suite: 'native', passed: false },
] }])).toEqual({ executed: 2, reused: 1, passed: 2, failed: 1, manual_accepted: 0, attempts: 4 });
// Paid evals never retry: a later pass never replaces an earlier failed record.
] }])).toEqual({ executed: 3, reused: 1, passed: 2, failed: 2, manual_accepted: 0, attempts: 4 });
});
});
+237
View File
@@ -0,0 +1,237 @@
/**
* Fail-open regression suite for the paid lane verdict. Synthetic slice
* artifacts go through the real `--report` CLI path (the command the workflow
* report jobs run), so a change to the gate cannot turn a real failure green
* without one of these cases going red. Landed before the panel-verdict gate
* change; every later gate change extends it.
*/
import { afterAll, beforeAll, describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { parseRunManifest, type PaidRunManifest, type SliceResult } from '../scripts/test-paid-shards';
import { stampTrialSeries } from '../scripts/eval-trial-series';
const ROOT = path.resolve(import.meta.dir, '..');
const RULE_A = 'test/skill-e2e-fail-open-alpha.test.ts';
const RULE_B = 'test/skill-e2e-fail-open-beta.test.ts';
let base: string;
beforeAll(() => { base = fs.mkdtempSync(path.join(os.tmpdir(), 'paid-fail-open-')); });
afterAll(() => { fs.rmSync(base, { recursive: true, force: true }); });
type Outcome = SliceResult['outcomes'][number];
const passed = (file: string, extra: Partial<Outcome> = {}): Outcome =>
({ files: [file], status: 'passed', exitCode: 0, elapsedMs: 1_000, executedTests: 1, skippedTests: 0, ...extra });
function manifest(entries: PaidRunManifest['entries'], sliceCount: number): PaidRunManifest {
return parseRunManifest(JSON.stringify({ version: 1, tier: 'periodic', evalsAll: true, sliceCount,
selectionReason: 'fail-open fixture', profile: 'full', selection: { e2e: null, judges: null }, entries }));
}
let caseCounter = 0;
function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record<string, unknown> = {}, env: NodeJS.ProcessEnv = {}) {
const dir = path.join(base, `case-${++caseCounter}`);
fs.mkdirSync(dir, { recursive: true });
fs.writeFileSync(path.join(dir, 'manifest.json'), JSON.stringify(plan));
for (const slice of slices) fs.writeFileSync(path.join(dir, `slice-${slice.sliceIndex}.json`), JSON.stringify(slice));
for (const [name, body] of Object.entries(collectors)) {
fs.mkdirSync(path.dirname(path.join(dir, name)), { recursive: true });
fs.writeFileSync(path.join(dir, name), JSON.stringify(body));
}
const result = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--tier', plan.tier, '--report', dir],
{ cwd: ROOT, encoding: 'utf8', timeout: 30_000, env: { ...process.env, GITHUB_RUN_ID: '', GITHUB_SHA: '', EVALS_TIER: plan.tier, ...env } });
return { status: result.status, out: `${result.stdout}\n${result.stderr}`, dir };
}
const slice = (sliceIndex: number, sliceCount: number, outcomes: Outcome[]): SliceResult =>
({ version: 1, tier: 'periodic', profile: 'full', selection: { e2e: null, judges: null }, sliceIndex, sliceCount, outcomes });
describe('rule shards stay fail-closed through --report', () => {
const plan = manifest([
{ file: RULE_A, slice: 1, status: 'planned' },
{ file: RULE_B, slice: 2, status: 'planned' },
], 2);
test('all planned rule shards passed: green', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A)]), slice(2, 2, [passed(RULE_B)])]);
expect(r.status, r.out).toBe(0);
});
test('a failed rule shard: red, naming the file', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'failed', exitCode: 1 })]), slice(2, 2, [passed(RULE_B)])]);
expect(r.status).toBe(1);
expect(r.out).toContain(`${RULE_A}: failed`);
});
test('a timed-out rule shard: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'timed-out', exitCode: null })]), slice(2, 2, [passed(RULE_B)])]);
expect(r.status).toBe(1);
expect(r.out).toContain(`${RULE_A}: timed-out`);
});
test('a missing slice artifact: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A)])]);
expect(r.status).toBe(1);
expect(r.out).toContain('slice 2/2 reported NO result');
});
test('a planned shard no slice reported: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A)]), slice(2, 2, [])]);
expect(r.status).toBe(1);
expect(r.out).toContain(`planned ${RULE_B} (slice 2) was never reported`);
});
test('a hollow shard under EVALS_ALL: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'passed-empty', executedTests: 0 })]), slice(2, 2, [passed(RULE_B)])]);
expect(r.status).toBe(1);
expect(r.out).toContain(`${RULE_A}: passed-empty`);
});
test('a never-started shard: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A, { status: 'never-started', exitCode: null, executedTests: null })]), slice(2, 2, [passed(RULE_B)])]);
expect(r.status).toBe(1);
});
test('a failed collector record under a passing shard: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A)]), slice(2, 2, [passed(RULE_B)])], {
'shards/skill-e2e-fail-open-alpha/run.json': { tier: 'e2e', total_tests: 1, total_cost_usd: 0,
tests: [{ name: 'alpha', suite: 's', tier: 'e2e', passed: false, duration_ms: 1, cost_usd: 0 }] },
});
expect(r.status).toBe(1);
expect(r.out).toContain('1 unapproved final collector failure(s)');
});
test('a shard reported by the wrong slice: red', () => {
const r = report(plan, [slice(1, 2, [passed(RULE_A), passed(RULE_B)]), slice(2, 2, [])]);
expect(r.status).toBe(1);
expect(r.out).toContain(`${RULE_B} planned for slice 2 but reported by slice 1`);
});
});
describe('behavior and quarantined panels through --report', () => {
const FILE = 'test/skill-e2e-review.test.ts';
const ID = 'review-design-lite';
const key = (trial: number) => `${FILE}#${ID}~t${trial}`;
const plan = (quarantined = false) => manifest([
...[1, 2, 3].map(trial => ({ file: key(trial), slice: trial, status: 'planned' as const,
trial: { kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined } })),
{ file: RULE_A, slice: 4, status: 'planned' },
], 4);
type TrialResult = 'passed' | 'failed' | 'contract' | 'missing' | 'harness';
const trialOutcome = (trial: number, result: TrialResult, quarantined = false): Outcome | null => {
if (result === 'missing') return null;
const record = { case: ID, trial, kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined, cost_usd: 0, duration_ms: 1_000 };
if (result === 'harness') return passed(key(trial), { status: 'never-started', exitCode: null, executedTests: null, skippedTests: null,
trial: { ...record, outcome: null, harness: 'never started' } });
if (result === 'passed') return passed(key(trial), { trial: { ...record, outcome: 'passed' } });
return passed(key(trial), { status: 'failed', exitCode: 1,
trial: { ...record, outcome: 'failed', failure_class: result === 'contract' ? 'contract' : 'timeout',
exit_reason: 'timeout', timeout_at_turn: 14, error: result === 'contract' ? 'handoff missing' : 'no posture match' } });
};
const run = (results: TrialResult[], quarantined = false, dropSlice?: number) => report(plan(quarantined), [1, 2, 3, 4]
.filter(index => index !== dropSlice)
.map(index => slice(index, 4, index === 4 ? [passed(RULE_A)]
: [trialOutcome(index, results[index - 1]!, quarantined)].filter((o): o is Outcome => o !== null))));
test('behavior 3/3: green', () => {
const r = run(['passed', 'passed', 'passed']);
expect(r.status, r.out).toBe(0);
expect(r.out).toContain('VERDICT GREEN');
});
test('behavior 2/3: green, the failed trial shown with its cause', () => {
const r = run(['passed', 'failed', 'passed']);
expect(r.status, r.out).toBe(0);
expect(r.out).toContain(`⚠ ${ID} behavior PASS 2/3 (✓✗✓)`);
expect(r.out).toContain('t2: timeout at turn 14');
const summary = JSON.parse(fs.readFileSync(path.join(r.dir, 'collector-outcomes.json'), 'utf8'));
expect(summary.version).toBe(2);
expect(summary.panels[0]).toMatchObject({ case: ID, status: 'PASS', split: true, failsLane: false });
const outcomesFile = path.join(r.dir, 'trial-outcomes.jsonl');
const history = () => fs.readFileSync(outcomesFile, 'utf8').trim().split('\n').map(line => JSON.parse(line));
expect(history().map(h => [h.trial, h.outcome])).toEqual([[1, 'passed'], [2, 'failed'], [3, 'passed']]);
expect(stampTrialSeries(outcomesFile)).toBe(3);
expect(new Set(history().map(h => h.series_identity)).size).toBe(1);
expect(history()[0].series_identity).toMatch(/^[0-9a-f]{16}$/);
});
test('behavior 1/3: red', () => {
const r = run(['passed', 'failed', 'failed']);
expect(r.status).toBe(1);
expect(r.out).toContain(`PANEL ${ID} FAIL 1/3`);
});
test('a missing trial record: INCOMPLETE, red', () => {
const r = run(['passed', 'missing', 'passed']);
expect(r.status).toBe(1);
expect(r.out).toContain(`PANEL ${ID} INCOMPLETE`);
});
test('a trial the harness never started: red, machine-classified for one re-dispatch', () => {
const r = run(['passed', 'harness', 'passed']);
expect(r.status).toBe(1);
expect(r.out).toContain('no trial record (never started)');
expect(r.out).toContain('INFRA-ONLY RED');
});
test('a contract trial at 2/3: red', () => {
const r = run(['passed', 'passed', 'contract']);
expect(r.status).toBe(1);
expect(r.out).toContain(`PANEL ${ID} FAIL 2/3`);
expect(r.out).toContain('contract violation');
expect(r.out).not.toContain('INFRA-ONLY RED');
});
test('quarantined 1/3: reported, does not fail the lane', () => {
const r = run(['passed', 'failed', 'failed'], true);
expect(r.status, r.out).toBe(0);
expect(r.out).toContain(`◌ ${ID} behavior (quarantined) FAIL 1/3`);
});
test('quarantined 0/3: hard break, red', () => {
const r = run(['failed', 'failed', 'failed'], true);
expect(r.status).toBe(1);
expect(r.out).toContain('quarantined hard break');
});
test('quarantined contract violation: red', () => {
const r = run(['passed', 'passed', 'contract'], true);
expect(r.status).toBe(1);
});
test('a missing trial slice: red', () => {
const r = run(['passed', 'passed', 'passed'], false, 2);
expect(r.status).toBe(1);
expect(r.out).toContain('slice 2/4 reported NO result');
expect(r.out).toContain(`PANEL ${ID} INCOMPLETE`);
});
test('a later run attempt never replaces the first attempt verdict', () => {
const r = run(['passed', 'failed', 'failed']);
expect(r.status).toBe(1);
const retry = slice(3, 4, [trialOutcome(3, 'passed')!]);
fs.mkdirSync(path.join(r.dir, 'paid-slice-3-a2'), { recursive: true });
fs.writeFileSync(path.join(r.dir, 'paid-slice-3-a2', 'slice-3.json'), JSON.stringify({ ...retry, attempt: 2 }));
const again = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--tier', 'periodic', '--report', r.dir],
{ cwd: ROOT, encoding: 'utf8', timeout: 30_000 });
expect(again.status).toBe(1);
expect(again.stdout).toContain('attempt 1 (later attempts 2 reported, never replacing it)');
});
test('verdicts become next-run receipts: a whole fresh PASS panel, and negatives for FAIL panels', () => {
const inputKey = 'b'.repeat(64);
const withKey = (results: TrialResult[]) => [1, 2, 3, 4].map(index => slice(index, 4, index === 4 ? [passed(RULE_A)]
: [{ ...trialOutcome(index, results[index - 1]!)!, inputKey }]));
const env = { GITHUB_RUN_ID: '77', GITHUB_SHA: 'e'.repeat(40) };
const green = report(plan(), withKey(['passed', 'failed', 'passed']), {}, env);
expect(green.status, green.out).toBe(0);
const receipt = JSON.parse(fs.readFileSync(path.join(green.dir, 'report-receipts', `${inputKey}.panel.json`), 'utf8'));
expect(receipt).toMatchObject({ key: inputKey, case: ID, panel: { n: 3, k: 2 }, source: { runId: '77/1' } });
expect(receipt.trials.map((t: any) => t.outcome)).toEqual(['passed', 'failed', 'passed']);
const red = report(plan(), withKey(['passed', 'failed', 'failed']), {}, env);
expect(red.status).toBe(1);
expect(fs.readdirSync(path.join(red.dir, 'report-receipts'))).toEqual([`${inputKey}.fail.json`]);
});
});
+30 -46
View File
@@ -7,7 +7,7 @@ import {
shardFile, sliceExecutionOrder, sliceSupervisedWallMs, CASE_SHARDED_FILES,
} from '../scripts/test-paid-shards';
import {
ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS, RETRY_MAX_CASE_MS, SHORT_CASE_RETRY_FILES,
ALL_TIERS, AUQ_CONSISTENCY_RETRY_BUDGET, FILE_RETRY_BUDGETS,
FINDING_RETRY_BUDGETS, STRICT_RETRY_CASE_BUDGETS,
} from './helpers/eval-budgets';
@@ -15,16 +15,16 @@ import { E2E_TOUCHFILES } from './helpers/touchfiles';
const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8');
const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file));
// Walls cover every attempt the retry rule allows: files with a case budget
// past RETRY_MAX_CASE_MS run once (a timed-out attempt is a verdict).
// Paid evals never retry (approved 2026-09-29): each wall covers one run of
// every case plus the supervision reserve.
const expectedWalls = {
'test/skill-e2e-qa-callers.test.ts': 3_270_000,
'test/skill-e2e-qa-callers.test.ts': 1_695_000,
'test/skill-e2e-shared-libs-paths.test.ts': 1_920_000,
'test/skill-e2e-ship-docsync.test.ts': 4_920_000,
'test/skill-llm-eval.test.ts': 6_220_000,
'test/skill-llm-eval.test.ts': 3_170_000,
'test/skill-e2e-auq-consistency.test.ts': 1_080_000,
'test/skill-e2e-auq-matrix.test.ts': 3_720_000,
'test/skill-e2e-plan-format.test.ts': 2_600_000,
'test/skill-e2e-auq-matrix.test.ts': 1_920_000,
'test/skill-e2e-plan-format.test.ts': 1_360_000,
'test/skill-e2e-auto-decide-preserved.test.ts': 1_020_000,
'test/skill-e2e-plan-ceo-finding-floor.test.ts': 1_020_000,
'test/skill-e2e-plan-eng-finding-floor.test.ts': 1_020_000,
@@ -33,38 +33,22 @@ const expectedWalls = {
'test/skill-e2e-plan-mode-no-op.test.ts': 3_120_000,
'test/skill-e2e-plan-ceo-mode-routing.test.ts': 1_320_000,
'test/skill-e2e-plan-eng-plan-mode.test.ts': 1_320_000,
'test/skill-e2e-plan-prosons.test.ts': 2_600_000,
'test/skill-e2e-plan-prosons.test.ts': 1_360_000,
'test/skill-e2e-plan.test.ts': 3_720_000,
};
test('retry rule: only files whose every case is CAPTURE tier or shorter retry; longer cases run once', () => {
expect(RETRY_MAX_CASE_MS).toBe(ALL_TIERS.CAPTURE_MS + 15_000);
test('paid evals never retry: every paid file and registered row runs once', () => {
for (const row of FILE_RETRY_BUDGETS) {
expect(row.retries, row.file).toBe(row.caseMs <= RETRY_MAX_CASE_MS ? (row.file.endsWith('plan-mode-no-op.test.ts') ? 2 : 1) : 0);
expect(retriesForFiles([row.file])).toBe(row.retries);
expect(Object.hasOwn(row, 'retries'), row.file).toBe(false);
expect(retriesForFiles([row.file])).toBe(0);
}
expect(FILE_RETRY_BUDGETS.filter(row => row.retries > 0).map(row => row.file).sort()).toEqual([
'test/skill-e2e-auq-matrix.test.ts', 'test/skill-e2e-plan-format.test.ts', 'test/skill-e2e-plan-prosons.test.ts',
'test/skill-e2e-qa-callers.test.ts', 'test/skill-llm-eval.test.ts',
]);
const paid = collectPaidTestFiles();
for (const file of SHORT_CASE_RETRY_FILES) {
expect(paid, `stale SHORT_CASE_RETRY_FILES entry: ${file}`).toContain(file);
expect(FILE_RETRY_BUDGETS.some(row => row.file === file)).toBe(false);
const source = read(file);
// Declared short budgets only: a JUDGE/CAPTURE tier or a literal at most the
// cap, no longer tier and no ms literal past the cap.
const literals = [...source.matchAll(/(?<![\w.])(\d{1,3}(?:_\d{3})+|\d{5,})(?![\w.])/g)]
.map(match => Number(match[1]!.replace(/_/g, '')));
expect(/\b(?:JUDGE_MS|CAPTURE_MS)\b/.test(source) || literals.some(ms => ms >= 60_000 && ms <= RETRY_MAX_CASE_MS), file).toBe(true);
expect(source, file).not.toMatch(/\b(?:CAPTURE_LONG_MS|PTY_MS|PTY_LONG_MS|OVERLAY_CASE_[A-Z_]+)\b/);
expect(literals.filter(ms => ms > RETRY_MAX_CASE_MS && ms < 10_000_000), file).toEqual([]);
expect(retriesForFiles([file])).toBe(1);
for (const file of collectPaidTestFiles()) expect(retriesForFiles([file]), file).toBe(0);
expect(buildPaidShardArgs(['test/x.test.ts'], 1000, 2)).toContain('--retry');
expect(buildPaidShardArgs(['test/x.test.ts'], 1000, 2).join(' ')).toContain('--retry 0');
const scripts: Record<string, string> = JSON.parse(read('package.json')).scripts;
for (const [name, command] of Object.entries(scripts)) {
if (/^test:(?:evals|e2e|gate|periodic)/.test(name)) expect(command, name).not.toMatch(/--retry(?:\s+|=)[1-9]/);
}
for (const file of paid.filter(file => !SHORT_CASE_RETRY_FILES.includes(file) && !FILE_RETRY_BUDGETS.some(row => row.file === file))) {
expect(retriesForFiles([file]), file).toBe(0);
}
expect(retriesForFiles([SHORT_CASE_RETRY_FILES[0]!, 'test/skill-e2e-plan.test.ts'])).toBe(0);
});
test('registration covers exactly the seventeen demonstrated full-file retry gaps', () => {
@@ -133,14 +117,14 @@ for (const row of newBudgets) {
outcomes: [{ files: [key], status: 'passed' as const, exitCode: 0, elapsedMs: 1, executedTests: count,
skippedTests: 0, budget: resolvePaidShardBudget([key]) }] }];
test(`${row.file}: full wall and existing retries propagate through planning`, () => {
expect(retriesForFiles([row.file])).toBe(row.retries);
test(`${row.file}: full wall propagates through planning and runs once`, () => {
expect(retriesForFiles([row.file])).toBe(0);
expect(resolvePaidShardBudget([row.file])).toEqual({ timeoutMs: expectedWalls[row.file as keyof typeof expectedWalls], source: 'registered', policyId: row.id });
expect(planPaidShards(['test/a.test.ts', row.file, 'test/z.test.ts'], { maxFilesPerShard: 3 })).toContainEqual([row.file]);
expect(() => resolvePaidShardBudget([row.file, 'test/neighbor.test.ts'])).toThrow('own shard');
expect(resolvePaidShardBudget([row.file], 50)).toEqual({ timeoutMs: 50, source: 'explicit', policyId: row.id });
expect(buildPaidShardArgs([row.file], row.shardMs, 2, retriesForFiles([row.file]))).toEqual([
'test', row.file, '--retry', String(row.retries), '--concurrent', '--max-concurrency=2', `--timeout=${row.shardMs}`,
'test', row.file, '--retry', '0', '--concurrent', '--max-concurrency=2', `--timeout=${row.shardMs}`,
]);
});
@@ -191,17 +175,17 @@ test('quality judge supervision includes the added judge without changing ordina
expect(ALL_TIERS).toEqual({ JUDGE_MS: 120000, CAPTURE_MS: 300000, CAPTURE_LONG_MS: 600000, PTY_MS: 900000, PTY_LONG_MS: 1200000 });
const quality = 'test/skill-llm-eval.test.ts';
const qualityBudget = FILE_RETRY_BUDGETS.find(row => row.file === quality)!;
expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 6_220_000, source: 'registered', policyId: qualityBudget.id });
expect(retriesForFiles([quality])).toBe(1);
expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 3_170_000, source: 'registered', policyId: qualityBudget.id });
expect(retriesForFiles([quality])).toBe(0);
const qualitySource = read(quality);
const judgeTimeouts = [...qualitySource.matchAll(/}\s*,\s*(JUDGE_MS|WORKFLOW_JUDGE_TEST_MS)\s*\);/g)].map(match => match[1]);
expect(judgeTimeouts.filter(timeout => timeout === 'JUDGE_MS')).toHaveLength(7);
expect(judgeTimeouts.filter(timeout => timeout === 'WORKFLOW_JUDGE_TEST_MS')).toHaveLength(17);
expect(qualitySource).toContain('WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000');
expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS');
expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000);
expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([
...Array(2).fill([1, 1500000, 0, 1620000]),
expect(qualityBudget.shardMs).toBe(7 * ALL_TIERS.JUDGE_MS + 17 * (ALL_TIERS.JUDGE_MS + 10_000) + 120_000);
expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.shardMs])).toEqual([
...Array(2).fill([1, 1500000, 1620000]),
]);
for (const tier of ['gate', 'periodic'] as const) {
const m = buildRunManifest({ tier, sliceCount: 1, evalsAll: true, env: { EVALS_ALL: '1' } });
@@ -227,7 +211,7 @@ test('detached PR fallback and release commands cover their actual default worke
const prFloor = Math.ceil((Math.ceil(fullGateFiles.length / prWorkers) * 1_800_000 + fullGateFiles.reduce(
(total, file) => total + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - 1_800_000), 0,
)) / 1000 * 1.05);
expect(prFloor).toBe(77_501);
expect(prFloor).toBe(72_755);
expect(prWall).toBe(92_820_000);
expect(prWall).toBeGreaterThanOrEqual(paidShardWallUpperBoundMs(files, prWorkers) + 120_000);
@@ -246,8 +230,8 @@ test('detached PR fallback and release commands cover their actual default worke
)) / 1000 * 1.05));
}
const detachedReleaseWall = Number(scripts['eval:bg:release'].match(/--timeout (\d+)/)?.[1]) * 1000;
expect(releaseFloors).toEqual([26_471, 30_797]);
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(57_268);
expect(releaseFloors).toEqual([21_725, 33_821]);
expect(releaseFloors.reduce((total, floor) => total + floor, 0)).toBe(55_546);
expect(detachedReleaseWall).toBe(116_700_000);
expect(detachedReleaseWall).toBeGreaterThanOrEqual(releaseWall + 120_000);
});
@@ -301,7 +285,7 @@ test('both gate executors plan the complete census and supervise every planned s
}
});
test('the periodic executor supervises every actual case and retry within its planned CI wall', () => {
test('the periodic executor supervises every actual case within its planned CI wall', () => {
const workflow: any = Bun.YAML.parse(read('.github/workflows/evals-periodic.yml'));
const executor = workflow.jobs['eval-slices'];
const emit = workflow.jobs['plan-slices'].steps.filter((step: any) =>
@@ -318,7 +302,7 @@ test('the periodic executor supervises every actual case and retry within its pl
evalsAll: true, env: { EVALS_ALL: '1' } });
const census = manifest.entries.filter(row => row.status === 'planned');
expect(new Set(census.map(row => shardFile(row.file)))).toEqual(new Set(selectPaidTestFiles(collectPaidTestFiles(), 'periodic').selected));
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_220_000);
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(3_170_000);
const walls = Array.from({ length: manifest.sliceCount }, (_, i) => sliceSupervisedWallMs(sliceExecutionOrder(
census.filter(row => row.slice === i + 1)).map(row => row.file), active.jobs));
expect(manifest.plan!.ciTimeoutMinutes * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
+120 -7
View File
@@ -37,11 +37,17 @@ import {
summarize,
summaryExitCode,
verifySliceResults,
expandTrialShards,
formatCapacityPreflight,
panelReports,
shardSlug,
type PaidRunManifest,
type ShardOutcome,
type SliceResult,
} from '../scripts/test-paid-shards';
import { E2E_KINDS } from './helpers/touchfiles-data';
const ROOT = path.resolve(__dirname, '..');
const outcome = (over: Partial<ShardOutcome>): ShardOutcome => ({
@@ -487,8 +493,8 @@ describe('hollow-shard guard', () => {
});
describe('retry parity', () => {
test('registered native workflows follow the retry rule while overlay attempts stay isolated', () => {
// A 25-minute case is past RETRY_MAX_CASE_MS: its timed-out attempt is the verdict.
test('registered native workflows and overlays run once', () => {
// Paid evals never retry: a timed-out attempt is the verdict.
const native = 'test/skill-e2e-plan-ceo-split-overflow.test.ts';
expect(retriesForFiles([native])).toBe(0);
expect(retriesForFiles([native.replaceAll('/', '\\')])).toBe(0);
@@ -496,16 +502,123 @@ describe('retry parity', () => {
const overlay = 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts';
expect(retriesForFiles([overlay])).toBe(0);
});
test('the matrix-era earned retries now follow the timeout-is-a-verdict rule, and each names a real file', () => {
// These three old matrix rows earned `retries: 2`; every one has a
// CAPTURE_LONG case, so a timed-out attempt is now their verdict.
test('the matrix-era earned retries are retired, and each names a real file', () => {
// These three old matrix rows earned `retries: 2`; paid evals never retry.
for (const file of ['test/skill-e2e-office-hours-auto-mode.test.ts', 'test/skill-e2e-plan-mode-no-op.test.ts', 'test/skill-e2e-workflow.test.ts']) {
expect(fs.existsSync(path.join(ROOT, file)), `stale retry parity entry: ${file}`).toBe(true);
expect(retriesForFiles([file])).toBe(0);
}
expect(retriesForFiles(['test/skill-e2e-retro.test.ts'])).toBe(0);
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(1);
expect(retriesForFiles(['test/skill-e2e-review.test.ts'])).toBe(0);
expect(buildPaidShardArgs(['x'], 1000, 4, 2)).toContain('2');
expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 1');
expect(buildPaidShardArgs(['x'], 1000, 4).join(' ')).toContain('--retry 0');
});
});
describe('trial planner (behavior and quarantined panels)', () => {
const REVIEW = 'test/skill-e2e-review.test.ts';
const budgetPlan = (tier: 'gate' | 'periodic', kinds: Record<string, 'rule' | 'behavior' | 'judge'>, quarantine: Record<string, unknown> = {}) =>
buildRunManifest({ tier, sliceBudgetMs: 540_000, jobs: 2, evalsAll: true, env: { EVALS_ALL: '1' },
kinds: { ...E2E_KINDS, ...kinds }, quarantine });
test('a behavior case becomes three trial shards on three different slices; its file shard runs the rest', () => {
const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' });
const trials = manifest.entries.filter(entry => entry.file.startsWith(`${REVIEW}#review-sql-injection~t`));
expect(trials.map(entry => entry.file)).toEqual([1, 2, 3].map(n => `${REVIEW}#review-sql-injection~t${n}`));
expect(trials.every(entry => entry.status === 'planned')).toBe(true);
expect(new Set(trials.map(entry => entry.slice)).size).toBe(3);
expect(trials[0]!.trial).toEqual({ kind: 'behavior', panel: { n: 3, k: 2 }, quarantined: false });
const fileShard = manifest.entries.find(entry => entry.file === REVIEW)!;
expect(fileShard.excludeCases).toEqual(['review-sql-injection']);
const slugs = manifest.entries.map(entry => shardSlug([entry.file]));
expect(new Set(slugs).size).toBe(slugs.length);
expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest);
});
test('a file whose only tier case is isolated drops its file shard', () => {
const manifest = budgetPlan('periodic', { 'review-design-lite': 'behavior' });
expect(manifest.entries.some(entry => entry.file === REVIEW)).toBe(false);
expect(manifest.entries.filter(entry => entry.file.startsWith(`${REVIEW}#`)).map(entry => entry.file))
.toEqual([1, 2, 3].map(n => `${REVIEW}#review-design-lite~t${n}`));
});
test('a quarantined rule case runs a full panel with k = n', () => {
const manifest = budgetPlan('gate', {}, { 'review-enum-completeness': { reason: 'r' } });
const trial = manifest.entries.find(entry => entry.file === `${REVIEW}#review-enum-completeness~t1`)!;
expect(trial.trial).toEqual({ kind: 'rule', panel: { n: 3, k: 3 }, quarantined: true });
});
test('slice-count plans keep trials on different slices too', () => {
const manifest = buildRunManifest({ tier: 'gate', sliceCount: 5, evalsAll: true, env: { EVALS_ALL: '1' },
kinds: { ...E2E_KINDS, 'review-sql-injection': 'behavior', 'review-enum-completeness': 'behavior' } });
for (const id of ['review-sql-injection', 'review-enum-completeness']) {
const slices = manifest.entries.filter(entry => entry.file.startsWith(`${REVIEW}#${id}~t`)).map(entry => entry.slice);
expect(new Set(slices).size).toBe(3);
}
});
test('judges and unknown ids cannot be isolated; unknown registrations throw', () => {
expect(() => budgetPlan('gate', { 'review/SKILL.md workflow': 'behavior' })).toThrow(/Only live E2E cases/);
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'trial-unknown-'));
try {
fs.mkdirSync(path.join(dir, 'test'));
fs.writeFileSync(path.join(dir, 'test/skill-e2e-x.test.ts'), 'const name = pick(); runSkillTest({ testName: name });');
expect(() => expandTrialShards(['test/skill-e2e-x.test.ts'], 'gate', dir, {
kinds: { x: 'behavior' }, touchfiles: { x: ['test/skill-e2e-x.test.ts'] }, tiers: { x: 'gate' },
})).toThrow(/statically known case registration/);
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
});
test('parse rejects partial panels, shared runners, forged plans and stray exclusions', () => {
const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' });
const trialFiles = [1, 2, 3].map(n => `${REVIEW}#review-sql-injection~t${n}`);
const mutate = (fn: (m: PaidRunManifest) => void) => { const m = structuredClone(manifest); fn(m); return JSON.stringify(m); };
expect(() => parseRunManifest(mutate(m => { m.entries = m.entries.filter(e => e.file !== trialFiles[1]); })))
.toThrow(/exactly its 3 trials/);
expect(() => parseRunManifest(mutate(m => {
const [a, b] = trialFiles.map(f => m.entries.find(e => e.file === f)!);
b!.slice = a!.slice;
}))).toThrow(/share a slice/);
expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === trialFiles[0])!.trial!.panel.k = 1; })))
.toThrow(/fixed policy panel/);
expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === trialFiles[0])!.trial = undefined; })))
.toThrow(/fixed policy panel/);
expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === REVIEW)!.excludeCases = ['review-design-lite']; })))
.toThrow(/exclude only cases/);
expect(() => parseRunManifest(mutate(m => { m.entries.find(e => e.file === REVIEW)!.trial = m.entries.find(e => e.file === trialFiles[0])!.trial; })))
.toThrow(/Only trial shards/);
});
test('capacity preflight names slices, shards, waves and the longest indivisible trial', () => {
const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' });
const lines = formatCapacityPreflight(manifest, 16).join('\n');
expect(lines).toContain(`${manifest.sliceCount} slice(s)`);
expect(lines).toContain('3 trial shard(s)');
expect(lines).toContain(`wave(s) at max-parallel 16: ${Math.ceil(manifest.sliceCount / 16)}`);
expect(lines).toMatch(/longest indivisible trial ~\d+\.\dm \(test\/skill-e2e-review\.test\.ts#review-sql-injection~t\d\)/);
});
test('durations: trials record their longest wall under the case key and seed their own estimate', () => {
const key = `${REVIEW}#review-sql-injection`;
const merged = mergePaidTestDurations({}, [{ version: 1, tier: 'gate', sliceIndex: 1, sliceCount: 1, outcomes: [1, 2, 3].map(n => ({
files: [`${key}~t${n}`], status: 'passed' as const, exitCode: 0, elapsedMs: n * 60_000, executedTests: 1, skippedTests: 0,
})) }]);
expect(merged).toEqual({ [key]: 180_000 });
const packed = packBySliceBudget([1, 2, 3].map(n => `${key}~t${n}`), 540_000, 2, merged);
expect(packed.slices).toHaveLength(3);
expect(Object.values(packed.estimates)).toEqual([180_000, 180_000, 180_000]);
});
test('reuse is whole-panel only: a panel mixing reused and fresh trials is INCOMPLETE', () => {
const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' });
const trials = manifest.entries.filter(entry => entry.trial);
const reused = { inputKey: 'c'.repeat(64), runId: '1001/1', revision: 'd'.repeat(40), completedAt: 1 };
const results = (reusedTrials: number[]): SliceResult[] => trials.map(entry => ({ version: 1, tier: 'gate', sliceIndex: entry.slice,
sliceCount: manifest.sliceCount, outcomes: [{ files: [entry.file], status: 'passed', exitCode: 0, elapsedMs: 1, executedTests: 1, skippedTests: 0,
trial: { case: 'review-sql-injection', trial: Number(entry.file.slice(-1)), ...entry.trial!, outcome: 'passed', cost_usd: 0, duration_ms: 1 },
...(reusedTrials.includes(Number(entry.file.slice(-1))) ? { reused } : {}) }] }));
expect(panelReports(manifest, results([]), 1)[0]).toMatchObject({ status: 'PASS' });
expect(panelReports(manifest, results([1, 2, 3]), 1)[0]).toMatchObject({ status: 'PASS' });
expect(panelReports(manifest, results([2]), 1)[0]).toMatchObject({ status: 'INCOMPLETE', failsLane: true, reason: expect.stringContaining('partial panel reuse') });
});
});
+122
View File
@@ -47,6 +47,17 @@ import {
selectPaidTestFiles,
buildRunManifest,
parseRunManifest,
classifyTrialShard,
sliceExitCode,
guardTrialRecords,
parseJUnitCases,
caseIdForTestName,
shardTrial,
excludedCasesNamePattern,
runCaseDiagnosis,
caseFile,
parseCliOptions,
type CaseTrialPlan,
type ShardOutcome,
} from '../scripts/test-paid-shards';
@@ -578,3 +589,114 @@ describe('all-skipped pass census', () => {
expect(reviewLine).not.toContain('SKIPPED');
});
});
describe('isolated trial shards: record, classification and slice exit', () => {
const plan: CaseTrialPlan = { kind: 'behavior', panel: { n: 3, k: 2 }, quarantined: false };
const key = (n: number) => `test/skill-e2e-review.test.ts#review-sql-injection~t${n}`;
const base = { status: 'passed' as const, executedTests: 1, skippedTests: 0, elapsedMs: 5 };
const none = { records: [], contract: null };
test('trial keys keep their case id, file and index', () => {
expect(shardCaseId(key(2))).toBe('review-sql-injection');
expect(shardFile(key(2))).toBe('test/skill-e2e-review.test.ts');
expect(shardTrial(key(2))).toBe(2);
expect(shardTrial('test/skill-e2e-review.test.ts#review-sql-injection')).toBeNull();
expect(shardSlug([key(2)])).toBe('skill-e2e-review--review-sql-injection.t2');
});
test('classification: verdicts versus harness problems', () => {
const c = (over: Partial<ShardOutcome>, evidence: { records: any[]; contract: string | null } = none) =>
classifyTrialShard({ ...base, ...over }, 'review-sql-injection', 1, plan, evidence);
expect(c({}).outcome).toBe('passed');
expect(c({}, { records: [], contract: 'handoff missing' })).toMatchObject({ outcome: 'failed', failure_class: 'contract', error: 'handoff missing' });
expect(c({ status: 'failed' }, { records: [{ passed: false, exit_reason: 'timeout', timeout_at_turn: 9, error: 'x' }], contract: null }))
.toMatchObject({ outcome: 'failed', failure_class: 'timeout', timeout_at_turn: 9 });
expect(c({ status: 'failed' }, { records: [{ passed: false, error: 'expected 3' }], contract: null })).toMatchObject({ outcome: 'failed', failure_class: 'assertion' });
expect(c({ status: 'timed-out', executedTests: null, skippedTests: null })).toMatchObject({ outcome: 'failed', failure_class: 'timeout' });
expect(c({ status: 'failed', executedTests: null, skippedTests: null })).toMatchObject({ outcome: 'failed', failure_class: 'infra' });
expect(c({ status: 'failed', executedTests: 0, skippedTests: 0 })).toMatchObject({ outcome: 'failed', failure_class: 'infra' });
expect(c({ executedTests: 1, skippedTests: 1 })).toMatchObject({ outcome: 'skipped' });
for (const over of [{ status: 'never-started' as const }, { status: 'passed-empty' as const }, { executedTests: 2 },
{ runnerError: 'spawn failed' }, { executedTests: 0, skippedTests: 0 }]) {
expect(c(over).outcome, JSON.stringify(over)).toBeNull();
}
});
test('slice exit: rule shards stay strict; failed trials never red the runner, missing records do', () => {
const trial = (outcome: 'passed' | 'failed' | null) => ({ status: outcome === 'failed' ? 'failed' as const : 'passed' as const,
trial: { case: 'c', trial: 1, ...plan, outcome, cost_usd: 0, duration_ms: 1, ...(outcome === null ? { harness: 'never started' } : {}) } });
expect(sliceExitCode([{ status: 'passed' }, trial('failed')])).toBe(0);
expect(sliceExitCode([{ status: 'failed' }, trial('passed')])).toBe(1);
expect(sliceExitCode([{ status: 'passed' }, trial(null)])).toBe(1);
expect(sliceExitCode([{ status: 'timed-out' }])).toBe(1);
const hollow = guardTrialRecords([{ ...trial('passed'), status: 'passed-empty' as const }]);
expect(hollow[0]!.trial!.outcome).toBeNull();
expect(sliceExitCode(hollow)).toBe(1);
});
test('runPaidShards binds each trial to its case, index and panel and records its outcome', async () => {
const evalDirBase = fs.mkdtempSync(path.join(os.tmpdir(), 'trial-shards-'));
try {
const script = (fail: boolean) => `const fs = require('fs'), path = require('path');
const dir = process.env.GSTACK_EVAL_DIR; fs.mkdirSync(dir, { recursive: true });
const env = Object.fromEntries(Object.entries(process.env).filter(([k]) => k.startsWith('GSTACK_EVAL_') || k === 'EVALS_SELECTION_JSON'));
fs.writeFileSync(path.join(dir, 'env.json'), JSON.stringify(env));
fs.writeFileSync(path.join(dir, 'run.json'), JSON.stringify({ tests: [{ name: 'review-sql-injection', passed: ${!fail}, cost_usd: 0.5,
duration_ms: 1, exit_reason: ${fail ? "'timeout'" : "'success'"}, timeout_at_turn: 4, model: 'm' }] }));
console.log("Ran 1 tests across 1 files. [1ms]"); process.exit(${fail ? 1 : 0});`;
const summary = await runPaidShards([[key(1)], [key(2)], [key(3)]], {
jobs: 3, evalDirBase, log: () => {}, trials: { [key(1)]: plan, [key(2)]: plan, [key(3)]: plan },
commandFor: files => ({ command: process.execPath, args: ['-e', script(files[0] === key(2))] }),
});
const byKey = (n: number) => summary.outcomes.find(o => o.files[0] === key(n))!;
expect(byKey(1).trial).toMatchObject({ case: 'review-sql-injection', trial: 1, outcome: 'passed', cost_usd: 0.5, model: 'm' });
expect(byKey(2).trial).toMatchObject({ trial: 2, outcome: 'failed', failure_class: 'timeout', exit_reason: 'timeout', timeout_at_turn: 4 });
expect(sliceExitCode(summary.outcomes)).toBe(0);
const env = JSON.parse(fs.readFileSync(path.join(evalDirBase, 'shards', shardSlug([key(3)]), 'env.json'), 'utf8'));
expect(env).toMatchObject({ GSTACK_EVAL_CASE_ID: 'review-sql-injection', GSTACK_EVAL_KIND: 'behavior', GSTACK_EVAL_TRIAL: '3',
GSTACK_EVAL_PANEL_N: '3', GSTACK_EVAL_PANEL_K: '2', GSTACK_EVAL_POLICY_VERSION: '1' });
expect(JSON.parse(env.EVALS_SELECTION_JSON).selected).toEqual(['review-sql-injection']);
} finally { fs.rmSync(evalDirBase, { recursive: true, force: true }); }
});
test('file shards exclude isolated names; JUnit cases map to registry ids or stay unattributed', () => {
const pattern = new RegExp(excludedCasesNamePattern(['review-sql-injection']));
expect(pattern.test('suite > review-sql-injection')).toBe(false);
expect(pattern.test('suite > review-enum-completeness')).toBe(true);
const cases = parseJUnitCases(`<testsuites><testsuite name="f">
<testcase name="review-sql-injection" classname="s" time="1.5" />
<testcase name="review-enum-completeness" classname="s" time="0.1"><failure type="TimeoutError" message="test &amp; timed out" /></testcase>
<testcase name="plain helper" classname="" time="0"><skipped /></testcase>
</testsuite></testsuites>`);
expect(cases).toEqual([
{ name: 'review-sql-injection', classname: 's', outcome: 'passed', timeMs: 1500 },
{ name: 'review-enum-completeness', classname: 's', outcome: 'failed', timeMs: 100, failureType: 'TimeoutError', message: 'test & timed out' },
{ name: 'plain helper', classname: '', outcome: 'skipped', timeMs: 0 },
]);
expect(caseIdForTestName('review-sql-injection')).toBe('review-sql-injection');
expect(caseIdForTestName(CASE_TEST_NAMES['plan-review-report']!)).toBe('plan-review-report');
expect(caseIdForTestName('plain helper')).toBeNull();
});
test('--case/--trials: local diagnosis flags are validated and never combine with CI modes', () => {
expect(parseCliOptions(['--case', 'review-sql-injection', '--trials', '5'], {})).toMatchObject({ caseId: 'review-sql-injection', trials: 5 });
expect(() => parseCliOptions(['--trials', '3'], {})).toThrow('--trials requires --case');
expect(() => parseCliOptions(['--case', 'no-such-case'], {})).toThrow('live E2E case id');
expect(() => parseCliOptions(['--case', 'review-sql-injection', '--report', '/tmp/r'], {})).toThrow('local diagnosis');
expect(caseFile('review-sql-injection')).toBe('test/skill-e2e-review.test.ts');
});
test('--case runs the CI panel runner and prints its panelVerdict', async () => {
const evalDirBase = fs.mkdtempSync(path.join(os.tmpdir(), 'case-diagnosis-'));
const lines: string[] = [];
try {
const verdict = await runCaseDiagnosis('review-sql-injection', { trials: 3, evalDirBase, log: line => lines.push(line), jobs: 3,
commandFor: files => ({ command: process.execPath, args: ['-e',
`console.log("Ran 1 tests across 1 files. [1ms]"); process.exit(${files[0]!.endsWith('~t3') ? 1 : 0});`] }) });
// A rule case keeps its meaning locally: every trial must pass.
expect(verdict).toMatchObject({ case: 'review-sql-injection', kind: 'rule', panel: { n: 3, k: 3 }, passed: 2, status: 'FAIL' });
expect(lines.join('\n')).toContain('--case review-sql-injection: 3 trial(s) of test/skill-e2e-review.test.ts');
expect(lines.join('\n')).toContain('FAIL 2/3 (✓✓✗)');
} finally { fs.rmSync(evalDirBase, { recursive: true, force: true }); }
});
});
+30 -2
View File
@@ -9,10 +9,11 @@ import { describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { CASE_CI_EXCLUDE, PERIODIC_CI_EXCLUDE } from './helpers/periodic-exclude-data';
import { CASE_CI_EXCLUDE, CASE_QUARANTINE, EVAL_POLICY, PERIODIC_CI_EXCLUDE } from './helpers/periodic-exclude-data';
import { E2E_TOUCHFILES } from './helpers/touchfiles';
import { quarantinePolicyProblems } from '../scripts/eval-flake-rank';
import { isPaidTestFile } from './helpers/paid-test-set';
import { buildRunManifest, CASE_SHARDED_FILES, expandCaseShards, partitionCaseExclusions, selectPaidTestFiles, shardCaseId, shardFile } from '../scripts/test-paid-shards';
import { buildRunManifest, CASE_SHARDED_FILES, expandCaseShards, fileCaseRegistration, partitionCaseExclusions, selectPaidTestFiles, shardCaseId, shardFile } from '../scripts/test-paid-shards';
const ROOT = path.resolve(__dirname, '..');
@@ -72,3 +73,30 @@ describe('periodic exclude policy', () => {
}
});
});
describe('eval verdict policy (pre-registered)', () => {
test('EVAL_POLICY carries exactly the approved constants; a change needs re-approval and a version bump', () => {
expect(EVAL_POLICY).toEqual({
version: 1,
panel: { n: 3, k: 2 },
quarantine: { entry: { rate: 0.95, minTrials: 10 }, exit: { rate: 0.97, minTrials: 10 }, capFraction: 0.10, expiryWeeklyRuns: 8 },
judge: { samples: 3 },
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
infraRedispatch: 1,
});
});
test('every CASE_QUARANTINE entry is a diagnosed, dated, non-product blocking case within the tier cap', () => {
expect(quarantinePolicyProblems(CASE_QUARANTINE).map(problem => problem.message)).toEqual([]);
});
test('a quarantined case runs as isolated trial shards: its files register it literally', () => {
for (const id of Object.keys(CASE_QUARANTINE)) {
const files = (E2E_TOUCHFILES[id] ?? []).filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file));
expect(files.length, `${id}: no paid file registers it`).toBeGreaterThan(0);
for (const file of files) {
expect(fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')).known, `${id}: ${file} registration must be statically known`).toBe(true);
}
}
});
});
+9 -11
View File
@@ -1,4 +1,4 @@
/** The real review registrations must finish capture cleanup before Bun retries. */
/** The real review registrations record late results and clean up before finalization, under the production zero-retry arguments. */
import { expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as os from 'node:os';
@@ -12,7 +12,7 @@ const CASES = [
['review-design-lite', 400, 35],
] as const;
for (const [id, workMs, maxTurns] of CASES) {
test.each(['recover', 'both-timeout'])(`${id} records late results before retry or finalization: %s`, scenario => {
test.each(['success', 'timeout'])(`${id} records late results before finalization: %s`, scenario => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'review-finalization-'));
const script = path.join(dir, 'registration.test.ts');
const facts = path.join(dir, 'events.jsonl');
@@ -41,7 +41,7 @@ mock.module(path.join(root, 'test/helpers/session-runner.ts'), () => ({
runSkillTest: async opts => {
const id = ++attempts;
event({ kind: 'start', id, timeout: opts.timeout, maxTurns: opts.maxTurns, cwd: opts.workingDirectory });
const timeout = id === 1 || ${JSON.stringify(scenario)} === 'both-timeout';
const timeout = ${JSON.stringify(scenario)} === 'timeout';
// Use the caller's actual work budget; only the provider and budget
// constants are scaled. The actual registered Bun outer deadline stays.
await new Promise(resolve => setTimeout(resolve, timeout ? opts.timeout + 50 : 80));
@@ -62,27 +62,25 @@ await import(path.join(root, ${JSON.stringify(PAID_FILE)}));
`);
try {
const retries = retriesForFiles([PAID_FILE]);
expect(retries).toBe(1);
expect(retries).toBe(0);
const child = Bun.spawnSync([process.execPath, ...buildPaidShardArgs([script], resolvePaidShardTimeoutMs([PAID_FILE]), 2, retries)], {
cwd: ROOT, timeout: 15_000, stdout: 'pipe', stderr: 'pipe',
env: { ...process.env, EVALS: '', EVALS_ALL: '', TMPDIR: dir, TMP: dir, TEMP: dir },
});
const output = child.stdout.toString() + child.stderr.toString();
expect(child.exitCode, output).toBe(scenario === 'recover' ? 0 : 1);
expect(child.exitCode, output).toBe(scenario === 'success' ? 0 : 1);
expect(output).not.toContain('Unhandled error between tests');
const events = fs.readFileSync(facts, 'utf8').trim().split('\n').map(line => JSON.parse(line));
const starts = events.filter(event => event.kind === 'start');
const ready = events.filter(event => event.kind === 'ready');
const records = events.filter(event => event.kind === 'record');
expect(starts.map(event => event.id)).toEqual([1, 2]);
expect(starts.map(event => event.id)).toEqual([1]);
expect(starts.map(({ timeout, maxTurns }) => ({ timeout, maxTurns })))
.toEqual([{ timeout: workMs, maxTurns }, { timeout: workMs, maxTurns }]);
expect(ready).toEqual([{ kind: 'ready', id: 1, fixtureExists: true }, { kind: 'ready', id: 2, fixtureExists: true }]);
.toEqual([{ timeout: workMs, maxTurns }]);
expect(ready).toEqual([{ kind: 'ready', id: 1, fixtureExists: true }]);
expect(records.map(event => [event.id, event.exitReason]))
.toEqual([[1, 'timeout'], [2, scenario === 'recover' ? 'success' : 'timeout']]);
.toEqual([[1, scenario === 'success' ? 'success' : 'timeout']]);
expect(events.findIndex(event => event.kind === 'record' && event.id === 1))
.toBeLessThan(events.findIndex(event => event.kind === 'start' && event.id === 2));
expect(events.findIndex(event => event.kind === 'record' && event.id === 2))
.toBeLessThan(events.findIndex(event => event.kind === 'finalized'));
expect(events.filter(event => event.kind === 'finalized')).toHaveLength(1);
expect(events.find(event => event.kind === 'registration')).toEqual({ kind: 'registration', name: id, outerMs: workMs + 50 + 5_000 });
+3 -3
View File
@@ -169,7 +169,7 @@ describe('setup-gbrain owned Path 4 fixture', () => {
pathToClaudeCodeExecutable: '/nonexistent/free-test-never-spawn-claude',
signal: controller.signal,
}, () => { validated = true; }, mode === 'deadline' ? 250 : 1000, {
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as EvalCollector,
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as unknown as EvalCollector,
name: 'setup-gbrain-path4-local-pglite', suite: 'setup-gbrain',
});
} catch (error) { failure = String(error); }
@@ -214,7 +214,7 @@ describe('setup-gbrain owned Path 4 fixture', () => {
await Promise.resolve();
controller.abort(new Error('caller cancelled during validation'));
}, 1000, {
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as EvalCollector,
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as unknown as EvalCollector,
name: 'setup-gbrain-path4-local-pglite', suite: 'setup-gbrain',
})).rejects.toThrow('caller cancelled during validation');
expect(rows).toHaveLength(1);
@@ -288,7 +288,7 @@ describe('setup-gbrain owned Path 4 fixture', () => {
throw new Error(`assertion diagnostic ${fixture.token} ${credentialUrl}`);
}
}, undefined, {
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as EvalCollector,
collector: { addTest: (row: EvalTestEntry) => rows.push(row) } as unknown as EvalCollector,
name: 'setup-gbrain-path4-local-pglite', suite: 'setup-gbrain',
});
} catch (error) { thrown = String(error); }
+2 -2
View File
@@ -13,9 +13,9 @@ import { DEFAULT_SHARD_TIMEOUT_MS, retriesForFiles } from '../scripts/test-paid-
const cases: ShipHookCase[] = ['ship-managed-hook-refresh', 'ship-unmanaged-hook-consent', 'ship-local-hook-preservation'];
type Fault = 'skip-guard' | 'skip-consent' | 'ask-overwrite' | 'direct-install' | 'read-receipts' | 'edit-policy' | 'tamper-receipts' | 'repeat-question' | 'rate-limit';
test('whole-file supervision covers every F5 case and the unchanged Bun retry', () => {
test('whole-file supervision covers every F5 case run once', () => {
for (const [file, count] of [['test/skill-e2e-ship-hook-refresh.test.ts', 1], ['test/skill-e2e-ship-hook-consent.test.ts', 2]] as const) {
expect(retriesForFiles([file])).toBe(1);
expect(retriesForFiles([file])).toBe(0);
expect(count * CAPTURE_MS * (retriesForFiles([file]) + 1) + 120000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS);
}
});
+2 -2
View File
@@ -152,9 +152,9 @@ function protocol(fault?: Fault, billing?: Array<number | undefined>, controls:
return { provider, directory: () => directory, calls: () => calls, sessions };
}
test('one bounded native case preserves the existing whole-file retry allowance', () => {
test('one bounded native case fits the whole-file wall and never retries', () => {
const retries = retriesForFiles(['test/skill-e2e-ship-skip.test.ts']);
expect(retries).toBe(1);
expect(retries).toBe(0);
expect(CAPTURE_MS * (retries + 1) + 120000).toBeLessThanOrEqual(DEFAULT_SHARD_TIMEOUT_MS);
});
+55 -46
View File
@@ -14,7 +14,7 @@ import { afterAll, expect } from 'bun:test';
import { JUDGE_MS } from './helpers/eval-budgets';
import * as fs from 'fs';
import * as path from 'path';
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelMajority, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS } from './helpers/llm-judge';
import { ENG_REVIEW_EXCERPT } from './helpers/workflow-excerpt';
import type { JudgeScore } from './helpers/llm-judge';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA, type WorkflowJudgeInput } from './helpers/workflow-judge-input';
@@ -100,8 +100,9 @@ describeIfSelected('LLM-as-judge quality evals', [
// rewrites the pin).
const section = sliceBrowseSection('## Snapshot Flags');
const scores = await judge('browse skill reference (flags + commands)', section);
console.log('Browse SKILL.md scores:', JSON.stringify(scores, null, 2));
const samples = await judgePanel(() => judge('browse skill reference (flags + commands)', section));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('Browse SKILL.md panel:', JSON.stringify({ mean: scores, samples }, null, 2));
const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json');
const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8'));
@@ -120,9 +121,9 @@ describeIfSelected('LLM-as-judge quality evals', [
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4 && regressions.length === 0,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: regressions.length ? `${scores.reasoning} | ${regressions.join('; ')}` : scores.reasoning,
judge_reasoning: regressions.length ? `${judgePanelReasoning(samples)} | ${regressions.join('; ')}` : judgePanelReasoning(samples),
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
@@ -144,8 +145,9 @@ describeIfSelected('LLM-as-judge quality evals', [
if (setupStart < 0 || setupEnd < 0) throw new Error('browse/SKILL.md: setup block not found — regenerate with: bun run gen:skill-docs');
const section = content.slice(setupStart, setupEnd);
const scores = await judge('setup/binary discovery instructions', section);
console.log('Setup block scores:', JSON.stringify(scores, null, 2));
const samples = await judgePanel(() => judge('setup/binary discovery instructions', section));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('Setup block panel:', JSON.stringify({ mean: scores, samples }, null, 2));
evalCollector?.addTest({
name: 'setup block',
@@ -153,9 +155,9 @@ describeIfSelected('LLM-as-judge quality evals', [
tier: 'llm-judge',
passed: scores.actionability >= 3 && scores.clarity >= 3,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
// Setup block is intentionally minimal (binary discovery only).
@@ -203,7 +205,7 @@ describeIfSelected('QA skill quality evals', ['qa/SKILL.md workflow', 'qa/SKILL.
startMarker: '# /qa: Test', endMarker: null,
references: ['qa/templates/functional-report-template.md'] }).text;
const scores = await callJudge<JudgeScore>(`You are evaluating the quality of a QA testing workflow document for an AI coding agent.
const samples = await judgePanel(() => callJudge<JudgeScore>(`You are evaluating the quality of a QA testing workflow document for an AI coding agent.
The agent reads this source-file bundle to select browser, native functional or mixed
surfaces, explore with bounded probes, reproduce and diagnose defects, add a regression
@@ -222,8 +224,9 @@ Respond with ONLY valid JSON:
Here is the QA workflow to evaluate:
${section}`);
console.log('QA workflow scores:', JSON.stringify(scores, null, 2));
${section}`));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('QA workflow panel:', JSON.stringify({ mean: scores, samples }, null, 2));
evalCollector?.addTest({
name: 'qa/SKILL.md workflow',
@@ -231,9 +234,9 @@ ${section}`);
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
@@ -247,7 +250,7 @@ ${section}`);
const t0 = Date.now();
const section = sliceQaPatterns('## Health Score Rubric');
const scores = await callJudge<JudgeScore>(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score.
const samples = await judgePanel(() => callJudge<JudgeScore>(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score.
The agent uses this rubric after QA testing a website. It needs to:
1. Understand each scoring category and what counts as a deduction
@@ -264,8 +267,9 @@ Respond with ONLY valid JSON:
Here is the rubric to evaluate:
${section}`);
console.log('QA health rubric scores:', JSON.stringify(scores, null, 2));
${section}`));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('QA health rubric panel:', JSON.stringify({ mean: scores, samples }, null, 2));
evalCollector?.addTest({
name: 'qa/SKILL.md health rubric',
@@ -273,9 +277,9 @@ ${section}`);
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
@@ -294,7 +298,7 @@ ${section}`);
const diffAwareSection = sliceQaPatterns('### Diff-aware', '### Full');
const rulesSection = sliceQaPatterns('## Important Rules');
const result = await callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario.
const samples = await judgePanel(() => callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario.
SCENARIO:
A user runs /qa (a browser-based QA testing skill). The branch diff shows ONLY prompt template files and config file changes — no routes, views, controllers, components, or CSS were changed. The changes are "purely backend" with no obvious UI surface.
@@ -318,9 +322,10 @@ Respond with ONLY valid JSON:
Rules:
- would_browse should be true if the document instructs the agent to always use the browser regardless of diff content
- would_browse should be false if the document allows the agent to skip browser testing for non-UI changes
- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`);
- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`));
const result = { would_browse: judgePanelMajority(samples, 'would_browse'), ...judgePanelMean(samples, ['confidence'] as const) };
console.log('QA anti-refusal result:', JSON.stringify(result, null, 2));
console.log('QA anti-refusal panel:', JSON.stringify({ result, samples }, null, 2));
evalCollector?.addTest({
name: 'qa/SKILL.md anti-refusal',
@@ -328,9 +333,9 @@ Rules:
tier: 'llm-judge',
passed: result.would_browse === true && result.confidence >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { would_browse: result.would_browse ? 1 : 0, confidence: result.confidence },
judge_reasoning: result.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(result.would_browse).toBe(true);
@@ -362,7 +367,7 @@ describeIfSelected('Cross-skill consistency evals', ['cross-skill greptile consi
extractGrepLines(retroContent, 'retro/SKILL.md'),
].join('\n\n');
const result = await callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently.
const samples = await judgePanel(() => callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently.
INTENDED ARCHITECTURE:
- greptile-history has TWO paths: per-project (~/.gstack/projects/{slug}/greptile-history.md) and global (~/.gstack/greptile-history.md)
@@ -383,9 +388,10 @@ Evaluate consistency. Respond with ONLY valid JSON:
"reasoning": "brief explanation"
}
score (1-5): 5 = perfectly consistent, 1 = contradictory`);
score (1-5): 5 = perfectly consistent, 1 = contradictory`));
const result = { consistent: judgePanelMajority(samples, 'consistent'), ...judgePanelMean(samples, ['score'] as const) };
console.log('Cross-skill consistency:', JSON.stringify(result, null, 2));
console.log('Cross-skill consistency panel:', JSON.stringify({ result, samples }, null, 2));
evalCollector?.addTest({
name: 'cross-skill greptile consistency',
@@ -393,9 +399,9 @@ score (1-5): 5 = perfectly consistent, 1 = contradictory`);
tier: 'llm-judge',
passed: result.consistent && result.score >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { consistency_score: result.score },
judge_reasoning: result.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(result.consistent).toBe(true);
@@ -439,7 +445,8 @@ async function runWorkflowJudge(opts: {
const workDeadline = started + JUDGE_MS;
let stage: 'input' | 'judge' | 'validation' | 'recording' = 'input';
let finalized = false;
let scores: JudgeScore | undefined;
let samples: JudgeScore[] | undefined;
let scores: Record<typeof JUDGE_SCORE_DIMENSIONS[number], number> | undefined;
let manualReview: ManualJudgeReview | undefined;
let customInputMetadata: { prompt: string; model: string } | undefined;
let reused: ReturnType<ReturnType<typeof prepareWorkflowJudgeCache>['lookup']> = null;
@@ -458,19 +465,19 @@ async function runWorkflowJudge(opts: {
evalCollector?.addTest({
name: opts.testName, suite: opts.suite, tier: 'llm-judge', passed, attempt,
duration_ms: Math.max(0, performance.now() - started),
cost_usd: reused || !scores ? 0 : 0.02,
cost_usd: reused || !samples ? 0 : 0.02 * samples.length,
execution: reused ? 'reused' : 'executed',
...customInputMetadata,
...(manualReview ? { manual_review: manualReview } : {}),
...(reused ? { reused_from: { input_key: reused.reuse.key, run_id: reused.reuse.source.runId,
revision: reused.reuse.source.revision, completed_at: new Date(reused.reuse.source.completedAt).toISOString() } } : {}),
...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning } : {}),
...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability } } : {}),
...(samples ? { judge_reasoning: judgePanelReasoning(samples) } : {}),
...(passed ? {} : { exit_reason: error instanceof JudgeRefusalError ? 'provider_refusal'
: error instanceof Error && error.name === 'WorkflowJudgeDeadline' ? 'timeout'
: error instanceof Error && error.name === 'WorkflowJudgeSuperseded' ? 'cancelled'
: stage === 'validation' ? 'validation_failed' : 'harness_error',
error: `${error instanceof Error ? error.message : String(error)}${scores ? '' : error instanceof JudgeRefusalError
error: `${error instanceof Error ? error.message : String(error)}${samples ? '' : error instanceof JudgeRefusalError
? '\nNo automated score; provider refusal usage retained when manually accepted; cost unavailable.'
: '\nNo completed model response; cost and usage unavailable.'}` }),
});
@@ -508,11 +515,11 @@ async function runWorkflowJudge(opts: {
checkActive();
stage = 'judge';
const maxTokens = opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS;
let result: JudgeScore;
let result: JudgeScore[];
try {
result = reused?.scores ?? await callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,
result = reused?.samples ?? await judgePanel(() => callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,
...(opts.stream ? { stream: true } : {}),
...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) });
...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) }));
} catch (error) {
checkActive();
if (error instanceof JudgeRefusalError && customInputMetadata) {
@@ -529,20 +536,21 @@ async function runWorkflowJudge(opts: {
throw error;
}
checkActive();
scores = result;
samples = result;
console.log(`[workflow-judge] ${opts.testName}: ${reused ? `reused ${reused.reuse.source.runId} @ ${reused.reuse.source.revision} (${new Date(reused.reuse.source.completedAt).toISOString()})` : 'executed'}`);
console.log(`${opts.testName} scores:`, JSON.stringify(scores, null, 2));
stage = 'validation';
if (opts.structuredResponse && !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true)) {
if (opts.structuredResponse && !samples.every(sample => validWorkflowJudgeScore(sample as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true))) {
throw new Error('Structured workflow judge violated the response schema');
}
scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log(`${opts.testName} panel:`, JSON.stringify({ mean: scores, samples }, null, 2));
expect(scores.clarity).toBeGreaterThanOrEqual(thresholds.clarity);
expect(scores.completeness).toBeGreaterThanOrEqual(thresholds.completeness);
expect(scores.actionability).toBeGreaterThanOrEqual(thresholds.actionability);
checkActive();
stage = 'recording';
arm();
const discardReceipt = reused ? undefined : cache.publish(scores, active);
const discardReceipt = reused ? undefined : cache.publish(samples, active);
try { checkActive(); finish(true); }
catch (error) { discardReceipt?.(); throw error; }
};
@@ -792,7 +800,7 @@ describeIfSelected('Voice directive eval', ['voice directive tone'], () => {
const voiceEnd = content.indexOf('\n## ', voiceStart + 1);
const voiceSection = content.slice(voiceStart, voiceEnd > 0 ? voiceEnd : voiceStart + 3000);
const result = await callJudge<{
const samples = await judgePanel(() => callJudge<{
directness: number;
concreteness: number;
avoids_corporate: number;
@@ -812,9 +820,10 @@ Return JSON only:
{"directness": N, "concreteness": N, "avoids_corporate": N, "avoids_ai_vocabulary": N, "connects_user_outcomes": N, "reasoning": "..."}
THE VOICE DIRECTIVE:
${voiceSection}`);
${voiceSection}`));
const result = judgePanelMean(samples, ['directness', 'concreteness', 'avoids_corporate', 'avoids_ai_vocabulary', 'connects_user_outcomes'] as const);
console.log('Voice directive scores:', JSON.stringify(result, null, 2));
console.log('Voice directive panel:', JSON.stringify({ mean: result, samples }, null, 2));
evalCollector?.addTest({
name: 'voice directive tone',
@@ -823,7 +832,7 @@ ${voiceSection}`);
passed: result.directness >= 4 && result.concreteness >= 4 && result.avoids_corporate >= 4
&& result.avoids_ai_vocabulary >= 4 && result.connects_user_outcomes >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: {
directness: result.directness,
concreteness: result.concreteness,
@@ -831,7 +840,7 @@ ${voiceSection}`);
avoids_ai_vocabulary: result.avoids_ai_vocabulary,
connects_user_outcomes: result.connects_user_outcomes,
},
judge_reasoning: result.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(result.directness).toBeGreaterThanOrEqual(4);
+2 -2
View File
@@ -292,12 +292,12 @@ test('F9 changed-input selection produces three cases with exact patterns and co
expect(prProfileTestNamePattern(files[1], selected.selection)).toBe('(?:^|\\s)(?:investigate-owned-abort|investigate-owned-ending-error)$');
});
test('both F9 files fit the existing wall with every Bun retry and reserve', () => {
test('both F9 files fit the existing wall with their one run and reserve', () => {
for (const file of files) {
const source = fs.readFileSync(path.join(import.meta.dir, '..', file), 'utf8');
const count = PR_PROFILE_FILES[file].length;
expect([...source.matchAll(/\}, CAPTURE_MS\);/g)]).toHaveLength(count);
expect(retriesForFiles([file])).toBe(1);
expect(retriesForFiles([file])).toBe(0);
const budget = resolvePaidShardBudget([file]);
expect(budget).toEqual({ timeoutMs: DEFAULT_SHARD_TIMEOUT_MS, source: 'default', policyId: null });
expect(count * CAPTURE_MS * (retriesForFiles([file]) + 1) + 120000).toBeLessThanOrEqual(budget.timeoutMs);
+146 -46
View File
@@ -1,18 +1,22 @@
import { afterEach, expect, spyOn, test } from 'bun:test';
import { afterEach, describe, expect, spyOn, test } from 'bun:test';
import { Messages } from '@anthropic-ai/sdk/resources/messages';
import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMajority, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge';
import { EVAL_POLICY } from './helpers/periodic-exclude-data';
import { getCookieWorkflowManualReview } from './helpers/cookie-workflow-manual-review';
import { resolveEvalModel } from '../lib/eval-model';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { execFileSync } from 'node:child_process';
import { prepareWorkflowJudgeCache, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
import { prepareWorkflowJudgeCache, validWorkflowJudgePanel, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA } from './helpers/workflow-judge-input';
const roots: string[] = [];
afterEach(() => { for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); });
const scores = { clarity: 4, completeness: 5, actionability: 4, reasoning: 'Concrete steps' };
const SAMPLES = JUDGE_PANEL_SAMPLES;
const panelOf = (sample: typeof scores) => Array.from({ length: SAMPLES }, () => sample);
const panel = panelOf(scores);
function fixture() {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-judge-cache-')); roots.push(root);
const files = {
@@ -53,9 +57,9 @@ function fixture() {
}
test('the audited adapter reuses only the exact completed score and original provenance', () => {
const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(scores);
const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(panel);
expect(f.entries()).toHaveLength(1);
const reused = f.cache().lookup(); expect(reused?.scores).toEqual(scores);
const reused = f.cache().lookup(); expect(reused?.samples).toEqual(panel);
expect(reused?.reuse.source.runId).toBe('free-cache-test');
expect(reused?.reuse.source.revision).toMatch(/^[a-f0-9]{40}$/);
expect(reused?.reuse.source.completedAt).toBeLessThanOrEqual(Date.now());
@@ -73,11 +77,11 @@ test('the dependency closure includes actual installed SDK bytes and local trans
});
test('release-label changes preserve reuse; other package semantics invalidate it', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
const file = path.join(f.root, 'package.json');
const original = JSON.parse(fs.readFileSync(file, 'utf8'));
fs.writeFileSync(file, JSON.stringify({ ...original, version: '2.0.0' }, null, 2));
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
for (const change of [{ scripts: { 'test:gate': 'changed command' } }, { dependencies: { 'some-sdk': '2.0.0' } }]) {
fs.writeFileSync(file, JSON.stringify({ ...original, ...change, version: '2.0.0' }));
expect(f.cache().lookup()).toBeNull();
@@ -89,7 +93,7 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in
'scripts/test-paid-shards.ts', 'scripts/test-strict-output.ts', 'scripts/eval-select.ts',
'scripts/test-pr-profile.ts', '.github/workflows/evals.yml']) {
test(`changes in ${file} require new evaluation`, () => {
const f = fixture(); f.cache().publish(scores); const target = path.join(f.root, file);
const f = fixture(); f.cache().publish(panel); const target = path.join(f.root, file);
fs.appendFileSync(target, file.endsWith('.json') ? ' ' : '\n// changed');
f.refreshPrompt(); expect(f.cache().lookup()).toBeNull();
});
@@ -98,8 +102,8 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in
test('changed sources during an attempt and mismatched actual prompt cannot publish', () => {
const f = fixture(); const before = f.cache();
fs.appendFileSync(path.join(f.root, 'example/sections/review.md'), 'new finding');
before.publish(scores); expect(f.entries()).toHaveLength(0);
f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(scores);
before.publish(panel); expect(f.entries()).toHaveLength(0);
f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(panel);
expect(f.entries()).toHaveLength(0);
});
@@ -108,52 +112,52 @@ for (const [key, value] of Object.entries({ EVALS_FRESH: '1', EVALS_TIER: 'perio
EVALS_CACHE_REPOSITORY: '', NODE_OPTIONS: '--require=unknown', BUN_OPTIONS: '--preload=unknown',
ANTHROPIC_BASE_URL: 'https://custom-provider.example.test' })) {
test(`${key}=${value} is fresh or ineligible`, () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.env = { ...f.env, [key]: value }; const cache = f.cache();
expect(cache.lookup()).toBeNull(); cache.publish(scores); expect(f.entries()).toHaveLength(1);
expect(cache.lookup()).toBeNull(); cache.publish(panel); expect(f.entries()).toHaveLength(1);
});
}
test('runtime/model/threshold changes miss, and retries never reuse or publish', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
for (const overrides of [{ GSTACK_EVAL_MODEL_JUDGE: 'different-model' }, { EVALS_CACHE_RUNTIME_ID: 'c'.repeat(64) }]) {
f.opts.env = { ...f.env, ...overrides }; expect(f.cache().lookup()).toBeNull();
}
f.opts.env = f.env; f.opts.thresholds.clarity = 5; expect(f.cache().lookup()).toBeNull();
f.opts.thresholds.clarity = 4; f.opts.attempt = 2; const retry = f.cache();
expect(retry.lookup()).toBeNull(); retry.publish(scores); expect(f.entries()).toHaveLength(1);
expect(retry.lookup()).toBeNull(); retry.publish(panel); expect(f.entries()).toHaveLength(1);
});
test('frontier reader calibration cannot reuse a score from the unspecified-reader rubric', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
const original = f.opts.prompt;
f.opts.agentCapability = 'frontier'; f.refreshPrompt();
expect(f.opts.prompt).not.toBe(original);
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(2);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
delete f.opts.agentCapability; f.refreshPrompt();
expect(f.opts.prompt).toBe(original);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('a pinned workflow judge model overrides the global model and changes the cache identity', () => {
const f = fixture();
f.opts.model = 'claude-sonnet-4-6';
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(1);
f.opts.env = { ...f.env, GSTACK_EVAL_MODEL_JUDGE: 'different-global-model' };
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
f.opts.model = 'claude-opus-4-7';
expect(f.cache().lookup()).toBeNull();
});
test('failed assertions, missing provenance, and missing imported dependencies cannot supply a receipt', () => {
const f = fixture(); f.cache().publish({ ...scores, clarity: 3 }); expect(f.entries()).toHaveLength(0);
f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(scores); expect(f.entries()).toHaveLength(0);
const f = fixture(); f.cache().publish(panelOf({ ...scores, clarity: 3 })); expect(f.entries()).toHaveLength(0);
f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(panel); expect(f.entries()).toHaveLength(0);
f.opts.env = f.env; fs.unlinkSync(path.join(f.root, 'test/helpers/nested.ts'));
f.cache().publish(scores); expect(f.entries()).toHaveLength(0);
f.cache().publish(panel); expect(f.entries()).toHaveLength(0);
});
test('cached payload schema remains small and cannot carry operational fields', () => {
@@ -167,8 +171,9 @@ test('workflow registration preserves model work and reserves only terminal-reco
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const body = source.split('async function runWorkflowJudge')[1]!.split('// Block 1:')[0]!;
const stages = ['workflowJudgeAttempts.set', 'readWorkflowJudgeInput(', 'cache.lookup()',
'callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(scores, active)']
'judgePanel(() => callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
'scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);',
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(samples, active)']
.map(stage => body.indexOf(stage));
expect(stages.every(position => position >= 0)).toBe(true);
expect(stages).toEqual([...stages].sort((a, b) => a - b));
@@ -203,6 +208,7 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
'evalCollector', 'expect', 'console', 'performance', 'JUDGE_MS', 'WORKFLOW_JUDGE_RECORD_MS',
'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'resolveEvalModel',
'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore',
'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS',
`${javascript}\nreturn runWorkflowJudge;`)(
f.root, overrides.read ?? readWorkflowJudgeInput, buildWorkflowJudgePrompt,
(options: WorkflowCacheOptions) => (overrides.prepare ?? prepareWorkflowJudgeCache)({ ...options, env: f.env }),
@@ -213,7 +219,8 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
overrides.clock ? { now: overrides.clock } : performance, overrides.budget ?? 120_000, overrides.allowance ?? 5_000,
overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout,
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore);
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore,
judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS);
return { run, records, signals, prompts, attempts, options: { ...f.opts, suite: 'Cache regression' } };
}
@@ -223,7 +230,7 @@ test('the actual workflow callback preserves the pinned model and frontier rubri
const actual = actualCallback(f, { judge: async (_prompt, model) => { models.push(model); return scores; } });
await actual.run({ ...actual.options, model: 'claude-sonnet-4-6', agentCapability: 'frontier',
readInput: () => readWorkflowJudgeInput(f.opts) });
expect(models).toEqual(['claude-sonnet-4-6']);
expect(models).toEqual(Array(SAMPLES).fill('claude-sonnet-4-6'));
expect(actual.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
expect(actual.records[0]).toMatchObject({ passed: true, model: 'claude-sonnet-4-6', prompt: actual.prompts[0] });
});
@@ -239,7 +246,7 @@ test.each(['ship', 'review'])('the registered %s callback sends the frontier rub
endMarker: f.opts.endMarker, references: [] };
const passing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 3 }) });
await passing.run(options);
expect(passing.prompts).toHaveLength(1);
expect(passing.prompts).toHaveLength(SAMPLES);
expect(passing.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
expect(passing.records[0]).toMatchObject({ passed: true, execution: 'executed', judge_scores: { clarity: 3 } });
const failing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 2 }) });
@@ -254,8 +261,8 @@ test('the actual workflow callback executes once, reuses with provenance, and pr
const f = fixture(); const first = actualCallback(f);
const options = { ...f.opts, suite: 'Cache regression' };
await first.run(options);
expect(first.prompts).toEqual([f.opts.prompt]); expect(f.entries()).toHaveLength(1);
expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 });
expect(first.prompts).toEqual(Array(SAMPLES).fill(f.opts.prompt)); expect(f.entries()).toHaveLength(1);
expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 * SAMPLES });
expect(first.records[0]).not.toHaveProperty('prompt');
expect(first.records[0]).not.toHaveProperty('model');
const reused = actualCallback(f, { judge: async () => ({ ...scores, clarity: 1 }) });
@@ -312,13 +319,13 @@ test('a superseding attempt cancels its predecessor before either can record a s
});
test('a failed input read consumes attempt one and prevents a retry from borrowing or publishing a receipt', async () => {
const f = fixture(); f.cache().publish(scores); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8');
const f = fixture(); f.cache().publish(panel); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8');
let reads = 0;
const h = actualCallback(f, { read: options => { if (++reads === 1) throw new Error('Missing workflow fixture'); return readWorkflowJudgeInput(options); } });
await expect(h.run(h.options)).rejects.toThrow('Missing workflow fixture');
expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'harness_error' });
await h.run(h.options);
expect(h.prompts).toHaveLength(1);
expect(h.prompts).toHaveLength(SAMPLES);
expect(h.records.map(record => record.execution)).toEqual(['executed', 'executed']);
expect(h.attempts.get(f.opts.testName).attempt).toBe(2);
expect(fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8')).toBe(receipt);
@@ -333,14 +340,14 @@ test('monotonic expiry after a synchronous preparation or late model response re
await expect(h.run(h.options)).rejects.toThrow('deadline');
expect(h.records).toHaveLength(1);
expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'timeout', duration_ms: 21 });
expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : 1);
expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : SAMPLES);
expect(f.entries()).toHaveLength(0);
}
});
test('publication rechecks after input scanning and withdraws a receipt if recording expires', async () => {
const f = fixture(); let checks = 0;
f.cache().publish(scores, () => ++checks < 2);
f.cache().publish(panel, () => ++checks < 2);
expect(checks).toBe(2); expect(f.entries()).toHaveLength(0);
let now = 0;
const h = actualCallback(f, { budget: 20, allowance: 5, clock: () => now,
@@ -373,7 +380,7 @@ test('the actual workflow callback preserves the complete public API body; cance
try {
const h = actualCallback(f, { judge: (prompt, model, options) => callJudge<typeof scores>(prompt, model, options) });
await h.run(h.options);
expect(create).toHaveBeenCalledTimes(1);
expect(create).toHaveBeenCalledTimes(SAMPLES);
expect(create.mock.calls[0]).toEqual([{
model: resolveEvalModel('judge'), max_tokens: 8192,
messages: [{ role: 'user', content: f.opts.prompt }],
@@ -412,40 +419,40 @@ test('Ship sends its authorized 64k cap and compact response contract through th
f.opts.structuredResponse = true;
f.opts.maxTokens = 65_536;
f.opts.stream = true;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
} finally { stream.mockRestore(); }
});
test('changing response serialization misses the cache even when prompt and model match', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.structuredResponse = true;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(2);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
const description = WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description;
try {
WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description += ' Changed response contract.';
expect(f.cache().lookup()).toBeNull();
} finally { WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description = description; }
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
f.opts.structuredResponse = false;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('the actual cap and streaming transport independently affect workflow cache identity', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.maxTokens = 65_536;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
f.opts.stream = true;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(3);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
delete f.opts.maxTokens;
delete f.opts.stream;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('the structured callback rejects incomplete, schema-invalid and below-threshold answers without cache credit', async () => {
@@ -473,3 +480,96 @@ test('the structured callback rejects incomplete, schema-invalid and below-thres
expect(validWorkflowJudgeScore({ ...scores, reasoning: Array(149).fill('word').join(' ') }, { clarity: 1, completeness: 1, actionability: 1 }, true)).toBe(true);
} finally { stream.mockRestore(); diagnostics.mockRestore(); }
});
// --- Judge panel policy (EVAL_POLICY.judge): fixed concurrent samples, per-dimension
// mean and boolean majority against unchanged thresholds, an erroring sample fails
// the whole panel and is never resampled. The provider is always a stub.
const panelScore = (clarity: number, completeness = 4, actionability = 4) => ({ clarity, completeness, actionability, reasoning: `c${clarity}` });
const panelThresholds = { clarity: 3, completeness: 3, actionability: 4 };
const panelRefusal = () => new JudgeRefusalError({ id: 'msg_1', _request_id: 'req_1', model: 'm', usage: { input_tokens: 1, output_tokens: 0 }, content: [] });
describe('judge panel', () => {
test('the pre-registered panel is three samples, and the helper restates EVAL_POLICY exactly', () => {
expect(EVAL_POLICY.judge.samples).toBe(3);
expect(JUDGE_PANEL_SAMPLES).toBe(EVAL_POLICY.judge.samples);
});
test('draws every sample concurrently before any resolves', async () => {
let started = 0;
const releases: Array<() => void> = [];
const panel = judgePanel(() => new Promise<number>(resolve => { started += 1; releases.push(() => resolve(started)); }));
await Promise.resolve();
expect(started).toBe(SAMPLES);
releases.forEach(release => release());
expect(await panel).toHaveLength(SAMPLES);
});
test('an erroring sample fails the panel and is never resampled', async () => {
let calls = 0;
const panel = judgePanel(async () => {
calls += 1;
if (calls === 2) throw new Error('Judge returned non-JSON: nope');
return panelScore(5);
});
await expect(panel).rejects.toThrow('non-JSON');
expect(calls).toBe(SAMPLES);
});
test('a refusal on every sample stays a provider refusal; a partial refusal is an ordinary failure', async () => {
await expect(judgePanel(async () => { throw panelRefusal(); })).rejects.toBeInstanceOf(JudgeRefusalError);
let calls = 0;
const partial = judgePanel(async () => { if (++calls === 1) throw panelRefusal(); return panelScore(4); });
const error = await partial.then(() => null, (reason: unknown) => reason);
expect(error).toBeInstanceOf(Error);
expect(error).not.toBeInstanceOf(JudgeRefusalError);
expect(String(error)).toContain(`sample 1 of ${SAMPLES} failed beside scored samples`);
});
test('numeric dimensions gate on the per-dimension mean; one low sample can be outvoted, a low mean cannot', () => {
const outvoted = judgePanelMean([panelScore(2), panelScore(4), panelScore(4)], JUDGE_SCORE_DIMENSIONS);
expect(outvoted.clarity).toBeCloseTo(10 / 3);
expect(outvoted.clarity).toBeGreaterThanOrEqual(panelThresholds.clarity);
const low = judgePanelMean([panelScore(2), panelScore(2), panelScore(4)], JUDGE_SCORE_DIMENSIONS);
expect(low.clarity).toBeLessThan(panelThresholds.clarity);
// No compensation across dimensions: each is averaged on its own.
expect(judgePanelMean([panelScore(5, 1), panelScore(5, 1), panelScore(5, 1)], JUDGE_SCORE_DIMENSIONS).completeness).toBe(1);
});
test('malformed sample fields fail the panel instead of averaging to NaN', () => {
expect(() => judgePanelMean([panelScore(4), { ...panelScore(4), clarity: '4' as unknown as number }, panelScore(4)], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2 has non-numeric clarity');
expect(() => judgePanelMean([panelScore(4), null as unknown as ReturnType<typeof panelScore>], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2');
expect(() => judgePanelMean([], JUDGE_SCORE_DIMENSIONS)).toThrow('no samples');
});
test('boolean fields gate on a strict majority', () => {
const vote = (...values: boolean[]) => judgePanelMajority(values.map(value => ({ ok: value })), 'ok');
expect(vote(true, true, false)).toBe(true);
expect(vote(true, false, false)).toBe(false);
expect(vote(true, false)).toBe(false);
expect(() => judgePanelMajority([{ ok: true }, { ok: 'yes' }], 'ok')).toThrow('sample 2 has non-boolean ok');
});
test('reasoning keeps every sample, numbered, even for malformed samples', () => {
expect(judgePanelReasoning([panelScore(4), null, { reasoning: 7 }])).toBe('[sample 1] c4\n[sample 2] \n[sample 3] ');
});
test('the cache stores and validates only a complete panel against the mean', () => {
expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(true);
expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(2), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), { ...panelScore(4), clarity: 6 }] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4)], prompt: 'x' }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel(panelScore(4), panelThresholds)).toBe(false);
});
test('every judge in the quality file samples through the panel, never a lone call', () => {
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const calls = [...source.matchAll(/\b(?:callJudge<[^>(]*(?:<[^>]*>[^>(]*)*>|judge)\(/g)];
expect(calls.length).toBeGreaterThanOrEqual(8);
for (const call of calls) {
expect(source.slice(Math.max(0, call.index! - 25), call.index), `unpaneled judge call at offset ${call.index}`).toMatch(/judgePanel\(\(\) => $/);
}
expect(source).not.toMatch(/\bscores\.reasoning\b|\bresult\.reasoning\b/);
});
});