From 3bae8e33da5beb9b1cb118ca92ed1e98e6682340 Mon Sep 17 00:00:00 2001 From: garrytan Date: Tue, 29 Sep 2026 19:41:39 +0000 Subject: [PATCH] feat(evals): planner-side whole-panel reuse and negative receipts The planner job restores this PR's receipt store once and ships a single filtered set with the plan: a pass or panel receipt with a same-or-newer FAIL for its input identity is dropped, and a panel receipt ships only as a whole PASS panel (re-verified with panelVerdict) from one run. Executors read only that set (no per-slice cache restore or save), so every trial of a panel sees the same receipts; a trial reuses its own record from the panel receipt, keeping a split PASS's failed trial. Trial identities drop the trial index (run-scoped) and bind the panel policy. Executed shards carry their input identity; the report turns a whole fresh PASS panel into a panel receipt and a FAIL panel or failed rule shard into a negative receipt, and marks a panel that mixes reused and fresh trials INCOMPLETE. The report job merges plan, slice and report receipts (newest per file) and saves one store per run. Also fixes two TS2352 casts in browse/test/dia-macos-qualification.test.ts whose diagnostic text drifted with program order (baseline locked, fix only). --- .github/workflows/evals.yml | 84 ++++++----- browse/test/dia-macos-qualification.test.ts | 4 +- scripts/e2e-shard-reuse.ts | 149 +++++++++++++++++++- scripts/test-paid-shards.ts | 73 ++++++++-- scripts/typecheck-test-baseline.json | 1 - test/ci-eval-cache.test.ts | 98 ++++++------- test/e2e-shard-reuse.test.ts | 78 +++++++++- test/paid-report-fail-open.test.ts | 19 ++- test/paid-run-manifest.test.ts | 14 ++ 9 files changed, 410 insertions(+), 110 deletions(-) diff --git a/.github/workflows/evals.yml b/.github/workflows/evals.yml index 7cf43e744..8dbbc99bd 100644 --- a/.github/workflows/evals.yml +++ b/.github/workflows/evals.yml @@ -140,10 +140,23 @@ jobs: with: bun-version: 1.4.0 + # Planner-side reuse: restore this PR's newest receipt store (the report + # job saves one merged store per run) and ship ONE filtered set with the + # plan, so every trial of a panel sees the same receipts and a newer FAIL + # blocks any older PASS for the same inputs. + - name: Restore this PR's verified judge and E2E results + if: github.event_name == 'pull_request' + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 + with: + path: /tmp/gstack-eval-input-cache + key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-plan + restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}- + - name: Emit run manifest if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all' env: EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }} + EVALS_CACHE_DIR: ${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }} run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 16 - name: Emit validation-phase manifest @@ -182,7 +195,9 @@ jobs: - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: paid-plan - path: /tmp/paid-plan/manifest.json + path: | + /tmp/paid-plan/manifest.json + /tmp/paid-plan/receipts retention-days: 30 eval-slices: @@ -251,15 +266,13 @@ jobs: name: paid-plan path: /tmp/paid-plan - # Only this PR's receipts are eligible. No base-branch or cross-PR restore - # prefix; every receipt also verifies exact inputs and its original age. - - name: Restore this PR's verified judge and E2E results - if: github.event_name == 'pull_request' - uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 - with: - path: /tmp/gstack-eval-input-cache - key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }} - restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}- + # Receipts come only from the plan (this PR's store, filtered once by the + # planner); new receipts land beside the slice results and the report + # merges them into the next store. + - name: Seed this slice's receipts from the plan + run: | + mkdir -p /tmp/paid-slice-results/receipts + if [ -d /tmp/paid-plan/receipts ]; then cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/; fi - name: Run slice ${{ matrix.slice }} env: @@ -270,35 +283,12 @@ jobs: EVALS_JOBS: "2" EVALS_CONCURRENCY: "2" GSTACK_EVAL_DIR: /tmp/paid-slice-results - EVALS_CACHE_DIR: /tmp/gstack-eval-input-cache + EVALS_CACHE_DIR: /tmp/paid-slice-results/receipts EVALS_CACHE_REPOSITORY: ${{ github.repository }} EVALS_CACHE_PR: ${{ github.event.pull_request.number }} EVALS_CACHE_RUNTIME_ID: ${{ needs.build-image.outputs.runtime-id }} run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/paid-plan/manifest.json --slice ${{ matrix.slice }} - - name: Find finalized passing receipts - id: receipts - if: ${{ !cancelled() && github.event_name == 'pull_request' }} - run: | - # Only a producer publishes. A later reuse-only slice must not become - # the newest prefix match and hide another slice's newly earned pass. - for receipt in /tmp/gstack-eval-input-cache/*.json; do - [ -f "$receipt" ] || continue - if jq -e --arg run "$GITHUB_RUN_ID/$GITHUB_RUN_ATTEMPT" '.proof.source.runId == $run' "$receipt" >/dev/null 2>&1; then - echo 'present=true' >> "$GITHUB_OUTPUT" - break - fi - done - - # An unrelated failing case does not discard already verified passes. - # Failed/retried/partial attempts never become receipts in the first place. - - name: Save verified judge and E2E results for this PR - if: ${{ !cancelled() && steps.receipts.outputs.present == 'true' }} - uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 - with: - path: /tmp/gstack-eval-input-cache - key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }} - # Attempt-scoped: a re-run attempt's trials are reported under that # attempt and never replace (or collide with) the first attempt's. - name: Upload slice results @@ -396,8 +386,27 @@ jobs: echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT" - name: Stamp trial history series - if: always() && hashFiles('/tmp/paid-report/trial-outcomes.jsonl') != '' - run: bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl + if: always() + run: | + if [ -f /tmp/paid-report/trial-outcomes.jsonl ]; then + bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl + fi + + # One merged receipt store per run: the plan's shipped set, every slice's + # new pass receipts, and the report's panel and negative receipts. Saved + # last, so the next planner restores it as the newest prefix match. + - name: Merge this run's receipts + if: always() && github.event_name == 'pull_request' + run: | + bun --no-install run scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache \ + /tmp/paid-report/receipts /tmp/paid-report/report-receipts /tmp/paid-report/paid-slice-*/receipts + + - name: Save this PR's verified judge and E2E results + if: always() && github.event_name == 'pull_request' + uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 + with: + path: /tmp/gstack-eval-input-cache + key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged - name: Upload reconciliation output for the comment job if: always() @@ -501,7 +510,8 @@ jobs: COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.' STATUS="✅ PASS" - if [ "${RECONCILE_EXIT:-1}" != "0" ]; then STATUS="❌ FAIL"; fi + if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ] \ + || { [ -n "$VERIFIED" ] && [ "$(jq -r '.verdict.verdict' "$VERIFIED")" != "GREEN" ]; }; then STATUS="❌ FAIL"; fi if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (verified report unavailable)'; fi diff --git a/browse/test/dia-macos-qualification.test.ts b/browse/test/dia-macos-qualification.test.ts index 40e13bf67..48e483158 100644 --- a/browse/test/dia-macos-qualification.test.ts +++ b/browse/test/dia-macos-qualification.test.ts @@ -1214,7 +1214,7 @@ Binary Images: { status: 1, stdout: '', stderr: '' }, { status: 0, stdout: '', stderr: '' }, { status: 0, stdout: 'truncated-private-row', stderr: '' }, { status: null, stdout: null, stderr: null, error: new Error('synthetic-private-error') }, - ]) expect(inspectUidProcesses(23456, performance.now() + 10_000, {}, (() => result) as typeof spawnSync)).toEqual({ available: false }); + ]) expect(inspectUidProcesses(23456, performance.now() + 10_000, {}, (() => result) as unknown as typeof spawnSync)).toEqual({ available: false }); }); test('numeric UID process filtering runs through the real global process table', () => { @@ -1251,7 +1251,7 @@ Binary Images: { status: 113, stdout: '', stderr: 'Could not find domain for user uid: 23456' }, { status: null, stdout: null, stderr: null, error: new Error('synthetic-private-error') }, ]) { - const observation = inspectUserDomain(23456, performance.now() + 10_000, {}, (() => result) as typeof spawnSync); + const observation = inspectUserDomain(23456, performance.now() + 10_000, {}, (() => result) as unknown as typeof spawnSync); expect(observation.state).toBe('unavailable'); expect(observation.structure).toBeUndefined(); expect(JSON.stringify(observation)).not.toContain('synthetic-private'); diff --git a/scripts/e2e-shard-reuse.ts b/scripts/e2e-shard-reuse.ts index 692d89250..f0a905a65 100644 --- a/scripts/e2e-shard-reuse.ts +++ b/scripts/e2e-shard-reuse.ts @@ -30,6 +30,8 @@ import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from './eval-input-cache'; import { matchGlob } from '../test/helpers/test-selection'; import { E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from '../test/helpers/touchfiles-data'; +import { EVAL_CACHE_MAX_AGE_MS as RECEIPT_MAX_AGE_MS } from './eval-input-cache'; +import { panelVerdict, TRIAL_ENV, type EvalCaseKind, type PanelShape, type PanelTrial } from '../test/helpers/eval-store'; export interface E2EShardReuseRequest { root: string; @@ -50,6 +52,8 @@ export interface E2EShardReuseRequest { profile: string; /** The exact environment the child receives. */ env: NodeJS.ProcessEnv; + /** Isolated trial shard: its panel policy is part of the identity; the trial index is run-scoped. */ + panel?: { kind: EvalCaseKind; panel: PanelShape; quarantined: boolean }; } export interface E2EShardReuseHit { key: string; source: EvalPassingProof['source'] } @@ -61,7 +65,7 @@ const HARNESS_FILES = ['scripts/test-paid-shards.ts', 'scripts/e2e-shard-reuse.t const ENV_PREFIXES = ['EVALS_', 'GSTACK_', 'CLAUDE_', 'ANTHROPIC_', 'OPENAI_', 'GEMINI_', 'BUN_', 'NODE_', 'PLAYWRIGHT_']; /** Run-scoped values: provenance or transport, never behavior. Selection is bound as case ids. */ const RUN_SCOPED_ENV = new Set(['EVALS_RUN_ID', 'GSTACK_EVAL_DIR', 'EVALS_CACHE_DIR', 'EVALS_CACHE_PR', 'EVALS_CACHE_REPOSITORY', - 'EVALS_CACHE_RUNTIME_ID', 'EVALS_CACHE_PURPOSE', 'EVALS_SELECTION_JSON', 'EVALS_JUDGE_SELECTION_JSON']); + 'EVALS_CACHE_RUNTIME_ID', 'EVALS_CACHE_PURPOSE', 'EVALS_SELECTION_JSON', 'EVALS_JUDGE_SELECTION_JSON', TRIAL_ENV.trial]); const SECRET_ENV = /KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL/; /** The reuse-relevant environment the child sees; secrets contribute presence only. */ @@ -126,7 +130,8 @@ export function e2eShardIdentity(request: E2EShardReuseRequest): { status: 'elig coverage: { dependencies: 'complete', prompts: 'complete', environment: 'complete' }, unknownDependencies: [], files, prompts: Object.fromEntries(request.caseIds.map(id => [id, source])), - parameters: { rootPackage, key: request.key, caseIds: [...request.caseIds].sort(), casePattern: request.casePattern, + parameters: { rootPackage, key: request.panel ? request.key.replace(/~t\d+$/, '') : request.key, + ...(request.panel ? { panel: { kind: request.panel.kind, n: request.panel.panel.n, k: request.panel.panel.k, quarantined: request.panel.quarantined } } : {}), caseIds: [...request.caseIds].sort(), casePattern: request.casePattern, expectedCases: request.expectedCases, retries: request.retries, timeoutMs: request.timeoutMs, withinShardConcurrency: request.withinShardConcurrency, tier: request.tier, profile: request.profile, environment: e2eReuseEnvironment(env) }, @@ -156,7 +161,13 @@ const validResult = (identity: EvalInputIdentity, key: string) => (value: EvalCa * whose inputs did not change during execution. */ export function prepareE2EShardReuse(request: E2EShardReuseRequest): { + /** The input identity key: recorded on the outcome so the report can store verdicts against it. */ + inputKey: string; lookup(): E2EShardReuseHit | null; + /** Trial shards only: this trial's record from a whole PASS panel receipt of the plan's receipts. */ + lookupPanelTrial(trial: number): { hit: E2EShardReuseHit; trial: PanelTrial } | null; + /** True when the inputs are unchanged since `before` (the outcome may carry inputKey). */ + unchanged(): boolean; publish(): void; } | null { if (e2eReuseLaneProblem(request.env, 'pr') !== null) return null; @@ -164,6 +175,19 @@ export function prepareE2EShardReuse(request: E2EShardReuseRequest): { if (before.status !== 'eligible') return null; const common = { cacheDir: request.env.EVALS_CACHE_DIR!, purpose: 'gate' as const }; return { + inputKey: before.identity.key, + unchanged() { + const after = e2eShardIdentity(request); + return after.status === 'eligible' && after.identity.key === before.identity.key; + }, + lookupPanelTrial(trial) { + if (!request.panel) return null; + const receipt = readPanelReceipt(common.cacheDir, before.identity.key); + if (!receipt || receipt.case !== request.caseIds[0] || receipt.kind !== request.panel.kind + || receipt.panel.n !== request.panel.panel.n || receipt.panel.k !== request.panel.panel.k) return null; + const record = receipt.trials.find(t => t.trial === trial); + return record ? { hit: { key: receipt.key, source: receipt.source }, trial: record } : null; + }, lookup() { const found = lookupEvalInputCache({ ...common, identity: before.identity, validateResult: validResult(before.identity, request.key) }); return found.status === 'reused' ? { key: found.key, source: found.source } : null; @@ -184,3 +208,124 @@ export function prepareE2EShardReuse(request: E2EShardReuseRequest): { }, }; } + +// ─── Panel receipts, negative receipts and the planner's receipt selection ── +// +// Reuse is decided by the planner, once per panel: it ships the plan a +// receipt set in which every panel receipt is a whole PASS panel from one run +// and no pass receipt has a newer FAIL for the same identity. Executors look +// up only that set, so every trial of a panel sees the same receipts. The +// report writes panel receipts (all n trials fresh, one identity) and +// negative receipts (FAIL verdicts) after the verdict is known. + +export interface PanelReceipt { + schema: 1; + key: string; + case: string; + kind: EvalCaseKind; + panel: PanelShape; + trials: PanelTrial[]; + source: { runId: string; revision: string; completedAt: number }; +} + +export interface NegativeReceipt { schema: 1; key: string; source: { runId: string; revision: string; completedAt: number } } + +const RECEIPT_KEY = /^[a-f0-9]{64}$/; +const validSource = (source: any) => !!source && typeof source.runId === 'string' && /^[\w./-]{1,160}$/.test(source.runId) + && typeof source.revision === 'string' && /^[a-f0-9]{40}$/.test(source.revision) && Number.isSafeInteger(source.completedAt) && source.completedAt > 0; + +function readJson(file: string, maxBytes = 64 * 1024): any { + try { + const stat = fs.lstatSync(file); + if (!stat.isFile() || stat.size > maxBytes) return null; + return JSON.parse(fs.readFileSync(file, 'utf8')); + } catch { return null; } +} + +/** A whole, unexpired PASS panel receipt for `key`, re-verified with panelVerdict(); else null. */ +export function readPanelReceipt(cacheDir: string, key: string, now = Date.now()): PanelReceipt | null { + if (!RECEIPT_KEY.test(key)) return null; + const receipt = readJson(path.join(cacheDir, `${key}.panel.json`)); + if (!receipt || receipt.schema !== 1 || receipt.key !== key || typeof receipt.case !== 'string' || !validSource(receipt.source) + || receipt.source.completedAt > now || now - receipt.source.completedAt >= RECEIPT_MAX_AGE_MS || !Array.isArray(receipt.trials)) return null; + try { + const verdict = panelVerdict({ case: receipt.case, kind: receipt.kind, panel: receipt.panel, + trials: receipt.trials.map((t: PanelTrial) => ({ ...t, attempt: 1 })) }); + if (verdict.status !== 'PASS' || verdict.trials.length !== receipt.panel.n) return null; + } catch { return null; } + const negative = readJson(path.join(cacheDir, `${key}.fail.json`)); + if (negative && validSource(negative.source) && negative.source.completedAt >= receipt.source.completedAt) return null; + return receipt as PanelReceipt; +} + +export function writePanelReceipt(dir: string, receipt: PanelReceipt): void { + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, `${receipt.key}.panel.json`), `${JSON.stringify(receipt)}\n`, { mode: 0o600 }); +} + +export function writeNegativeReceipt(dir: string, receipt: NegativeReceipt): void { + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, `${receipt.key}.fail.json`), `${JSON.stringify(receipt)}\n`, { mode: 0o600 }); +} + +const receiptTime = (file: string): number => { + const parsed = readJson(file); + return Number(parsed?.source?.completedAt ?? parsed?.proof?.source?.completedAt) || 0; +}; + +/** + * Planner-side selection: copy `from` into `to`, dropping every pass or panel + * receipt that has a same-or-newer negative receipt for its identity, and + * every panel receipt that is not a whole PASS panel. Workflow-judge and + * other receipts pass through for their own validation at lookup. + */ +export function selectPlanReceipts(from: string, to: string, now = Date.now()): { shipped: number; blocked: string[] } { + fs.mkdirSync(to, { recursive: true }); + const blocked: string[] = []; + let shipped = 0; + let names: string[] = []; + try { names = fs.readdirSync(from).filter(name => name.endsWith('.json')); } catch { return { shipped, blocked }; } + for (const name of names) { + const file = path.join(from, name); + const [key, suffix] = [name.slice(0, 64), name.slice(64)]; + const negative = RECEIPT_KEY.test(key) ? readJson(path.join(from, `${key}.fail.json`)) : null; + const newerFail = negative && validSource(negative.source) && negative.source.completedAt >= receiptTime(file); + if (suffix === '.panel.json' && (newerFail || !readPanelReceipt(from, key, now))) { blocked.push(name); continue; } + if (suffix === '.json' && newerFail) { blocked.push(name); continue; } + fs.copyFileSync(file, path.join(to, name)); + shipped++; + } + return { shipped, blocked }; +} + +/** Merge receipt directories into one store, keeping the newest file per name. */ +export function mergeReceiptDirs(out: string, dirs: string[]): number { + fs.mkdirSync(out, { recursive: true }); + let merged = 0; + for (const dir of dirs) { + let names: string[] = []; + try { names = fs.readdirSync(dir).filter(name => name.endsWith('.json')); } catch { continue; } + for (const name of names) { + const source = path.join(dir, name); + const target = path.join(out, name); + if (!fs.lstatSync(source).isFile()) continue; + if (fs.existsSync(target) && receiptTime(target) >= receiptTime(source)) continue; + fs.copyFileSync(source, target); + merged++; + } + } + return merged; +} + +if (import.meta.main) { + const [command, first, ...rest] = process.argv.slice(2); + if (command === 'select' && first && rest[0]) { + const result = selectPlanReceipts(first, rest[0]); + console.log(`[e2e-reuse] shipped ${result.shipped} receipt(s) to the plan; blocked ${result.blocked.length} (newer FAIL or partial panel)`); + } else if (command === 'merge' && first) { + console.log(`[e2e-reuse] merged ${mergeReceiptDirs(first, rest)} receipt(s) into ${first}`); + } else { + console.error('usage: bun run scripts/e2e-shard-reuse.ts select | merge '); + process.exit(2); + } +} diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 198f8aceb..1a5d9c8b4 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -78,7 +78,7 @@ import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-revi import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight'; import { OVERLAY_MIN_FILE_WALL_MS } from '../test/helpers/overlay-case-policy'; import { PR_PROFILE_CASE_IDS, PR_PROFILE_FILES, packageChangeOnlyVersion, selectPrProfile, type PrProfileSelection } from './test-pr-profile'; -import { e2eReuseLaneProblem, prepareE2EShardReuse } from './e2e-shard-reuse'; +import { e2eReuseLaneProblem, prepareE2EShardReuse, selectPlanReceipts, writeNegativeReceipt, writePanelReceipt } from './e2e-shard-reuse'; type E2EShardReuse = NonNullable>; import { @@ -826,6 +826,8 @@ export interface ShardOutcome { reused?: { inputKey: string; runId: string; revision: string; completedAt: number }; /** The parent could not run the shard at all (a runner error, never a trial verdict). */ runnerError?: string; + /** PR lane: the reuse input identity of a freshly executed shard whose inputs stayed unchanged. */ + inputKey?: string; /** Isolated trial shards only: the trial record this shard produced. */ trial?: ShardTrialRecord; } @@ -1057,7 +1059,22 @@ export async function runPaidShard( // Bootstrap-retention qualification binds per-run state, so that shard stays fresh. const reuse = files.some(file => normalizeRelativePath(file) === 'test/skill-e2e-qa-workflow.test.ts') ? null : options.reuseFor?.(files, env, budget) ?? null; - const reused = reuse?.lookup() ?? null; + // A trial reuses only its record from a whole PASS panel receipt the + // planner shipped; a single trial never has a pass receipt of its own. + const panelHit = trialPlan && trialIndex !== null ? reuse?.lookupPanelTrial(trialIndex) ?? null : null; + if (panelHit && trialPlan && trialIndex !== null && caseId !== null) { + const reusedFrom = { inputKey: panelHit.hit.key, runId: panelHit.hit.source.runId, revision: panelHit.hit.source.revision, completedAt: panelHit.hit.source.completedAt }; + const passedTrial = panelHit.trial.outcome === 'passed'; + log(`${label} REUSED trial ${trialIndex}/${trialPlan.panel.n} of ${caseId} (${panelHit.trial.outcome}) from the whole PASS panel of run ${reusedFrom.runId}`); + return { shard: shardNumber, files, status: passedTrial ? 'passed' : 'failed', exitCode: passedTrial ? 0 : 1, elapsedMs: 0, groupPid: null, + executedTests: 1, skippedTests: 0, budget, reused: reusedFrom, + trial: { case: caseId, trial: trialIndex, kind: trialPlan.kind, panel: trialPlan.panel, quarantined: trialPlan.quarantined, + outcome: panelHit.trial.outcome, cost_usd: 0, duration_ms: 0, + ...(panelHit.trial.failure_class ? { failure_class: panelHit.trial.failure_class } : {}), + ...(panelHit.trial.exit_reason ? { exit_reason: panelHit.trial.exit_reason } : {}), + ...(panelHit.trial.error ? { error: panelHit.trial.error } : {}) } }; + } + const reused = trialPlan ? null : reuse?.lookup() ?? null; if (reused) { const reusedFrom = { input_key: reused.key, run_id: reused.source.runId, revision: reused.source.revision, completed_at: new Date(reused.source.completedAt).toISOString() }; @@ -1255,7 +1272,8 @@ export async function runPaidShard( } } const elapsedMs = Date.now() - startedAt; - if (status === 'passed' && reuse) reuse.publish(); + if (status === 'passed' && reuse && !trialPlan) reuse.publish(); + const inputKey = reuse?.unchanged() ? reuse.inputKey : undefined; // Failure debuggability without the RAM cost: read back only the log's // tail. Live mode already streamed everything, so no re-print there. @@ -1273,7 +1291,8 @@ export async function runPaidShard( ? summary.terminalTestCounts.reduce((a, b) => a + b, 0) : null; const skippedTests = summary.terminalTestCounts.length > 0 ? summary.skippedTests : null; - return withTrial({ shard: shardNumber, files, status, exitCode, elapsedMs, groupPid, executedTests, skippedTests, budget }); + return withTrial({ shard: shardNumber, files, status, exitCode, elapsedMs, groupPid, executedTests, skippedTests, budget, + ...(inputKey ? { inputKey } : {}) }); } export interface RunSummary { @@ -2027,7 +2046,7 @@ export interface SliceResult { /** Epoch ms bounds of the slice's shard execution (lane wall time). */ startedAt?: number; finishedAt?: number; - outcomes: Array>; + outcomes: Array>; } /** @@ -2113,11 +2132,13 @@ export function verifySliceResults( if (outcome.reused !== undefined) { const r = outcome.reused; if (manifest.prCoverage?.mode !== 'pr') problems.push(`${file}: only the fast PR profile may reuse results; this lane executes fresh`); - if (outcome.status !== 'passed' || outcome.exitCode !== 0 || !/^[a-f0-9]{64}$/.test(r?.inputKey ?? '') + const reusedVerdictOk = outcome.trial ? outcome.trial.outcome !== null : outcome.status === 'passed' && outcome.exitCode === 0; + if (!reusedVerdictOk || !/^[a-f0-9]{64}$/.test(r?.inputKey ?? '') || !/^[\w./-]{1,160}$/.test(r?.runId ?? '') || !/^[a-f0-9]{40}$/.test(r?.revision ?? '') || !Number.isSafeInteger(r?.completedAt) || r.completedAt <= 0) { problems.push(`${file}: malformed reused result`); } } + if (outcome.inputKey !== undefined && !/^[a-f0-9]{64}$/.test(outcome.inputKey)) problems.push(`${file}: malformed input identity`); if (reported.has(file)) problems.push(`${file} reported by two slices`); reported.set(file, { slice: result.sliceIndex, status: outcome.status, ...(outcome.trial ? { trial: outcome.trial } : {}) }); const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === shardFile(file)); @@ -2301,6 +2322,12 @@ export function panelReports(manifest: PaidRunManifest, results: SliceResult[], ...(t.timeout_at_turn !== undefined ? { timeout_at_turn: t.timeout_at_turn } : {}) }]; }); const verdict = panelVerdict({ case: shardCaseId(key)!, kind: plan.kind, panel: plan.panel, trials, quarantined: plan.quarantined }); + // Reuse is whole-panel only: every trial reused from one run, or none. + const sources = entries.map(entry => reported.get(normalizeRelativePath(entry.file))?.outcome.reused?.runId ?? null); + if (sources.some(source => source !== null) && (sources.some(source => source === null) || new Set(sources).size > 1)) { + return { ...verdict, status: 'INCOMPLETE', split: false, failsLane: true, coverage: false, redClass: 'INCOMPLETE', + reason: 'partial panel reuse (every trial must come from one reused panel, or none)', file: shardFile(key), slices }; + } return { ...verdict, file: shardFile(key), slices }; }); } @@ -2507,6 +2534,29 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo const laterPanels = attempts.slice(1).flatMap(attempt => panelReports(manifest, artifacts.map(a => a.result), attempt) .filter(panel => panel.trials.length > 0)); + // PR-lane receipts from verdicts (the planner ships them to the next run): + // a whole fresh PASS panel with one input identity becomes a panel receipt; + // a FAIL panel or a failed rule shard becomes a negative receipt that + // blocks reuse of any older PASS for the same identity. + const revision = env.GITHUB_SHA ?? ''; + if (env.GITHUB_RUN_ID && /^[a-f0-9]{40}$/.test(revision)) { + const receiptsDir = path.join(reportDir, 'report-receipts'); + const source = { runId: `${env.GITHUB_RUN_ID}/${primary}`, revision, completedAt: Date.now() }; + const outcomes = results.flatMap(r => r.outcomes); + for (const panel of panels) { + const keys = outcomes.filter(o => o.trial?.case === panel.case && shardFile(o.files[0] ?? '') === panel.file).map(o => o.reused ? null : o.inputKey ?? null); + if (keys.length !== panel.panel.n || keys.some(k => k === null) || new Set(keys).size !== 1) continue; + if (panel.status === 'PASS') { + writePanelReceipt(receiptsDir, { schema: 1, key: keys[0]!, case: panel.case, kind: panel.kind, panel: panel.panel, source, + trials: panel.trials.map(({ trial, outcome, failure_class, exit_reason, error }) => ({ trial, outcome, + ...(failure_class ? { failure_class } : {}), ...(exit_reason ? { exit_reason } : {}), ...(error ? { error } : {}) })) }); + } else if (panel.status === 'FAIL') writeNegativeReceipt(receiptsDir, { schema: 1, key: keys[0]!, source }); + } + for (const outcome of outcomes.filter(o => !o.trial && o.inputKey && !o.reused && o.status !== 'passed')) { + writeNegativeReceipt(receiptsDir, { schema: 1, key: outcome.inputKey!, source }); + } + } + // Quarantine cap and expiry are the weekly pass-rates gate's (eval-flake-rank --gate). // History: one trial-outcomes line per isolated trial and per JUnit rule/judge case. @@ -2796,6 +2846,12 @@ async function main(): Promise { ); for (const line of formatSlicePlan(manifest)) console.log(line); for (const line of formatCapacityPreflight(manifest, options.maxParallel ?? undefined)) console.log(line); + // Planner-side reuse: ship ONE filtered receipt set with the plan, so + // every trial of a panel (on any slice) sees the same receipts. + if (manifest.profile === 'pr' && process.env.EVALS_CACHE_DIR) { + const shipped = selectPlanReceipts(process.env.EVALS_CACHE_DIR, path.join(path.dirname(path.resolve(options.emitPlanPath)), 'receipts')); + console.log(`[test:paid] reuse: shipped ${shipped.shipped} receipt(s) with the plan; blocked ${shipped.blocked.length} (newer FAIL or partial panel)`); + } return 0; } @@ -2865,6 +2921,7 @@ async function main(): Promise { const { registered, known } = fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')); const exclude = mine.find(entry => entry.file === key)?.excludeCases; return prepareE2EShardReuse({ root: ROOT, key, file, caseIds: prProfileShardIds(key, manifest.selection!, exclude), + ...(trials[normalizeRelativePath(key)] ? { panel: trials[normalizeRelativePath(key)] } : {}), registeredIds: registered, registrationKnown: known, casePattern: prProfileTestNamePattern(key, manifest.selection!, exclude), expectedCases: expectedPrCaseCount(key, manifest.selection!, exclude), retries: retriesForFiles(files), timeoutMs: budget.timeoutMs, withinShardConcurrency: options.withinShardConcurrency, @@ -2901,9 +2958,9 @@ async function main(): Promise { attempt: Number.isSafeInteger(attempt) && attempt > 0 ? attempt : 1, startedAt, finishedAt: Date.now(), - outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused, runnerError, trial }) => + outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused, runnerError, trial, inputKey }) => ({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}), ...(reused ? { reused } : {}), - ...(runnerError !== undefined ? { runnerError } : {}), ...(trial ? { trial } : {}) })), + ...(runnerError !== undefined ? { runnerError } : {}), ...(trial ? { trial } : {}), ...(inputKey ? { inputKey } : {}) })), }; fs.mkdirSync(evalDirBase, { recursive: true }); const sliceResultPath = path.join(evalDirBase, `slice-${options.sliceIndex}.json`); diff --git a/scripts/typecheck-test-baseline.json b/scripts/typecheck-test-baseline.json index 612942cf7..c96077b9e 100644 --- a/scripts/typecheck-test-baseline.json +++ b/scripts/typecheck-test-baseline.json @@ -36,7 +36,6 @@ "browse/test/cookie-import-transport.test.ts\tTS2741\tProperty 'preconnect' is missing in type '() => Promise' but required in type 'typeof fetch'.": 1, "browse/test/dia-gui-readiness.test.ts\tTS2352\tConversion of type '() => { status: number; stdout: string; stderr: string; }' to type '{ (command: string): SpawnSyncReturns; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ status: number; stdout: string; stderr: string; }' is missing the following properties from type 'SpawnSyncReturns': pid, output, signal": 1, "browse/test/dia-launch-comparison.test.ts\tTS7016\tCould not find a declaration file for module '../../.github/scripts/dia-launch-driver.mjs'. '.github/scripts/dia-launch-driver.mjs' implicitly has an 'any' type.": 1, - "browse/test/dia-macos-qualification.test.ts\tTS2352\tConversion of type '() => { error?: undefined; status: number; stdout: string; stderr: string; } | { status: null; stdout: null; stderr: null; error: Error; }' to type '{ (command: string): SpawnSyncReturns; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ error?: undefined; status: number; stdout: string; stderr: string; } | { status: null; stdout: null; stderr: null; error: Error; }' is not comparable to type 'SpawnSyncReturns'. Type '{ status: null; stdout: null; stderr: null; error: Error; }' is missing the following properties from type 'SpawnSyncReturns': pid, output, signal": 2, "browse/test/dia-macos-qualification.test.ts\tTS2352\tConversion of type '() => { status: number; stdout: string; stderr: string; }' to type '{ (command: string): SpawnSyncReturns; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ status: number; stdout: string; stderr: string; }' is missing the following properties from type 'SpawnSyncReturns': pid, output, signal": 2, "browse/test/dia-macos-qualification.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type 'string' is not assignable to parameter of type '\"browser_profile_unavailable\" | \"code_signing_error\" | \"debugging_pipe_unavailable\" | \"default_profile_policy\" | \"dynamic_library_error\" | \"graphics_or_bootstrap_error\" | \"keychain_access_failed\" | \"keychain_interaction_disallowed\" | \"keychain_interaction_required\"'.": 1, "browse/test/domain-skills-e2e.test.ts\tTS2339\tProperty 'cleanup' does not exist on type 'BrowserManager'.": 1, diff --git a/test/ci-eval-cache.test.ts b/test/ci-eval-cache.test.ts index a34f430f5..ce2fe5618 100644 --- a/test/ci-eval-cache.test.ts +++ b/test/ci-eval-cache.test.ts @@ -21,19 +21,31 @@ test('only PR runs select the fast profile; manual and scheduled coverage stays } }); -test('receipt transport restores only this repository and PR with no broad fallback key', () => { - const steps = paid.jobs['eval-slices'].steps; - const restore = steps.filter((s: any) => s.uses?.startsWith('actions/cache/restore@')); - const save = steps.filter((s: any) => s.uses?.startsWith('actions/cache/save@')); +test('receipt transport: the planner restores only this repository and PR, the report saves one merged store', () => { + const planner = paid.jobs['plan-slices'].steps; + const restore = planner.filter((s: any) => s.uses?.startsWith('actions/cache/restore@')); expect(restore).toHaveLength(1); - expect(save).toHaveLength(1); expect(restore[0].if).toBe("github.event_name == 'pull_request'"); + expect(restore[0].with.path).toBe('/tmp/gstack-eval-input-cache'); expect(restore[0].with['restore-keys']).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-'); - expect(save[0].with.key).toBe(restore[0].with.key); - expect(save[0].with.key).toContain('${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}'); + const emit = planner.find((s: any) => s.run?.includes('--emit-plan /tmp/paid-plan/manifest.json')); + expect(emit.env.EVALS_CACHE_DIR).toBe("${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }}"); + const upload = planner.find((s: any) => s.with?.name === 'paid-plan'); + expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/paid-plan/manifest.json', '/tmp/paid-plan/receipts']); + // Executors never restore or save a cache of their own: every slice sees the plan's one receipt set. + const executor = paid.jobs['eval-slices'].steps; + expect(executor.filter((s: any) => s.uses?.startsWith('actions/cache/'))).toHaveLength(0); + expect(executor.find((s: any) => s.name === "Seed this slice's receipts from the plan").run).toContain('cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/'); + const report = paid.jobs['slices-report'].steps; + const merge = report.find((s: any) => s.name === "Merge this run's receipts"); + expect(merge.run).toContain('scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache'); + const save = report.filter((s: any) => s.uses?.startsWith('actions/cache/save@')); + expect(save).toHaveLength(1); expect(save[0].with.path).toBe('/tmp/gstack-eval-input-cache'); - expect(save[0].if).toContain("steps.receipts.outputs.present == 'true'"); + expect(save[0].with.key).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged'); + expect(report.indexOf(save[0])).toBeGreaterThan(report.indexOf(merge)); expect(paid.jobs['eval-slices'].permissions).toEqual({ contents: 'read', packages: 'read' }); + expect(paid.jobs['slices-report'].permissions).toEqual({ contents: 'read' }); expect(JSON.stringify(periodic)).not.toContain('actions/cache/'); }); @@ -43,56 +55,21 @@ test('the judge binds cache receipts to the PR and installed runtime, not the co expect(runtime.run).toContain('sha256sum /tmp/eval-runtime-manifest.json'); const run = paid.jobs['eval-slices'].steps.find((s: any) => s.run?.includes('--plan /tmp/paid-plan/manifest.json')); expect(run.env).toMatchObject({ - EVALS_CACHE_DIR: '/tmp/gstack-eval-input-cache', + EVALS_CACHE_DIR: '/tmp/paid-slice-results/receipts', EVALS_CACHE_REPOSITORY: '${{ github.repository }}', EVALS_CACHE_PR: '${{ github.event.pull_request.number }}', EVALS_CACHE_RUNTIME_ID: '${{ needs.build-image.outputs.runtime-id }}', }); }); -test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('only a new passing producer can publish the next cache snapshot', () => { - const directory = mkdtempSync(join(tmpdir(), 'ci-cache-producer-')); - const receipts = join(directory, 'receipts'); - const output = join(directory, 'output'); - mkdirSync(receipts); - const step = paid.jobs['eval-slices'].steps.find((s: any) => s.id === 'receipts'); - const script = step.run.replaceAll('/tmp/gstack-eval-input-cache', receipts); - const run = () => { - writeFileSync(output, ''); - const result = spawnSync('bash', ['-e', '-c', script], { - env: { ...process.env, GITHUB_OUTPUT: output, GITHUB_RUN_ID: '42', GITHUB_RUN_ATTEMPT: '2' }, - encoding: 'utf8', timeout: 5000, - }); - expect(result.status, result.stderr).toBe(0); - return readFileSync(output, 'utf8'); - }; - try { - expect(run()).toBe(''); - writeFileSync(join(receipts, 'old.json'), JSON.stringify({ proof: { source: { runId: '41/1' } } })); - writeFileSync(join(receipts, 'corrupt.json'), '{'); - expect(run()).toBe(''); - writeFileSync(join(receipts, 'prior-attempt.json'), JSON.stringify({ proof: { source: { runId: '42/1' } } })); - expect(run()).toBe(''); - writeFileSync(join(receipts, 'fresh.json'), JSON.stringify({ proof: { source: { runId: '42/2' } } })); - expect(run()).toBe('present=true\n'); - } finally { rmSync(directory, { recursive: true, force: true }); } -}); - -test.skipIf(!Bun.which('jq'))('the actual comment separates reused evidence, retry outcomes and deferred coverage', () => { +test.skipIf(!Bun.which('jq'))('the actual comment shows deferred coverage and never recomputes a verdict', () => { const comment = paid.jobs['slices-comment'].steps.find((s: any) => s.name === 'Post PR comment').run as string; const evaluate = (filter: string, value: unknown) => { const result = spawnSync('jq', ['-r', filter], { input: JSON.stringify(value), encoding: 'utf8', timeout: 5000 }); expect(result.status, result.stderr).toBe(0); return result.stdout.trim(); }; - const stats = comment.match(/STATS=\$\(jq -r '([^']+)'/)![1]!; - expect(evaluate(stats, { tests: [ - { name: 'retry', passed: false }, { name: 'retry', passed: true }, - { name: 'exhausted', passed: false }, { name: 'exhausted', passed: false }, - { name: 'regressed', passed: true }, { name: 'regressed', passed: false }, - { name: 'reused', passed: true, execution: 'reused' }, - ], flaky_retries: ['retry', 'exhausted', 'regressed'].map(name => ({ name, attempts: 2 })) })).toBe('4 2 2 3 3 1'); - expect(comment).toContain("printf ' | ⚠ %s cases with multiple attempts'"); + expect(comment).not.toContain('group_by(.name)'); expect(comment).not.toMatch(/flaky pass\(es\)|passed only on retry|not blocking/); const coverage = comment.match(/COVERAGE=\$\(jq -r '([^']+)'/)![1]!; const text = evaluate(coverage, { profile: 'pr', selection: { e2e: ['probe'], judges: ['judge'] }, @@ -107,9 +84,9 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f const job = paid.jobs['slices-comment']; expect(job.permissions).toMatchObject({ 'pull-requests': 'write' }); expect(JSON.stringify(job.steps)).not.toMatch(/actions\/checkout|setup-bun|bun run|npm |node /); - const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict'); - expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json']); - expect(job.steps.find((step: any) => step.with?.name === 'report-verdict').with.path).toBe('/tmp/verdict'); + const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}'); + expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json', '/tmp/paid-report/report-summary.md']); + expect(job.steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}').with.path).toBe('/tmp/verdict'); const root = mkdtempSync(join(tmpdir(), 'ci-comment-')); const paidDir = join(root, 'paid-report'); const verdictDir = join(root, 'verdict'); @@ -123,9 +100,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f writeFileSync(join(paidDir, 'judge.json'), JSON.stringify({ total_tests: 2, tier: 'llm-judge', shard: 1, tests: [{ name: 'manual', passed: false, manual_review: { unverified: true } }, { name: 'reused', passed: true, execution: 'reused' }], flaky_retries: [] })); - const summary = { version: 1, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0, + const summary = { version: 2, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0, total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 }], - totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 } }; + totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 }, + verdict: { verdict: 'GREEN' }, headline: ['[test:paid] VERDICT GREEN — lane gate/pr, attempt 1'], panels: [], + failures: ['⚠ case-x behavior PASS 2/3 (✓✗✓) t2: timeout at turn 3 — @\u200bsomeone said no'] }; mkdirSync(join(verdictDir, 'paid-report')); const summaryPath = join(verdictDir, 'paid-report/collector-outcomes.json'); const script = (job.steps.find((step: any) => step.name === 'Post PR comment').run as string) @@ -142,8 +121,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f const verified = run(); expect(verified.status, verified.stderr).toBe(0); expect(verified.stdout).toContain('⚠ MANUAL ACCEPTED (unscored)'); - expect(verified.stdout).toContain('1 automated passed / 2 final results'); - expect(verified.stdout).toContain('0 failed, 1 manual accepted'); + expect(verified.stdout).toContain('VERDICT GREEN — lane gate/pr, attempt 1'); + expect(verified.stdout).toContain('1 executed, 1 reused** rule/judge records'); + expect(verified.stdout).toContain('1 manual accepted'); + expect(verified.stdout).toContain('### Failures and split verdicts'); + expect(verified.stdout).toContain('PASS 2/3 (✓✗✓) t2: timeout at turn 3'); const unrelatedFailure = { ...summary, files: [{ ...summary.files[0], total: 3, failed: 1, executed: 2, attempts: 3 }], totals: { ...summary.totals, total: 3, failed: 1, @@ -152,16 +134,20 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f const red = run(); expect(red.status, red.stderr).toBe(0); expect(red.stdout).toContain('❌ FAIL'); - expect(red.stdout).toContain('1 failed, 1 manual accepted'); + + writeFileSync(summaryPath, JSON.stringify({ ...summary, verdict: { verdict: 'RED' } })); + const redVerdict = run(); + expect(redVerdict.status, redVerdict.stderr).toBe(0); + expect(redVerdict.stdout).toContain('❌ FAIL'); writeFileSync(summaryPath, JSON.stringify({ ...summary, totals: { ...summary.totals, manual_accepted: 2 } })); const tampered = run(); expect(tampered.status, tampered.stderr).toBe(0); - expect(tampered.stdout).toContain('manual acceptance unavailable/unverified'); + expect(tampered.stdout).toContain('verified report unavailable'); expect(tampered.stdout).not.toContain('⚠ MANUAL ACCEPTED (unscored)'); rmSync(summaryPath); const absent = run(); expect(absent.status, absent.stderr).toBe(0); - expect(absent.stdout).toContain('manual acceptance unavailable/unverified'); + expect(absent.stdout).toContain('verified report unavailable'); } finally { rmSync(root, { recursive: true, force: true }); } }); diff --git a/test/e2e-shard-reuse.test.ts b/test/e2e-shard-reuse.test.ts index a07b8f8e5..acb515321 100644 --- a/test/e2e-shard-reuse.test.ts +++ b/test/e2e-shard-reuse.test.ts @@ -4,7 +4,8 @@ import * as os from 'node:os'; import * as path from 'node:path'; import { e2eReuseEnvironment, e2eReuseLaneProblem, e2eShardIdentity, e2eShardInputFiles, prepareE2EShardReuse, - type E2EShardReuseRequest, + mergeReceiptDirs, readPanelReceipt, selectPlanReceipts, writeNegativeReceipt, writePanelReceipt, + type E2EShardReuseRequest, type PanelReceipt, } from '../scripts/e2e-shard-reuse'; import { buildRunManifest, fileCaseRegistration, runPaidShard, verifySliceResults, type SliceResult } from '../scripts/test-paid-shards'; @@ -127,10 +128,13 @@ describe('E2E shard reuse through the runner', () => { test('a failed shard never publishes a receipt', async () => { let published = 0; const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, env: laneEnv(), log: () => {}, - reuseFor: () => ({ lookup: () => null, publish: () => { published++; } }), + reuseFor: () => ({ inputKey: 'e'.repeat(64), unchanged: () => true, lookupPanelTrial: () => null, + lookup: () => null, publish: () => { published++; } }), commandFor: () => ({ command: process.execPath, args: ['-e', 'process.exit(1)'] }) }); expect(outcome.status).toBe('failed'); expect(published).toBe(0); + // The identity rides on the outcome so the report can store the FAIL as a negative receipt. + expect(outcome.inputKey).toBe('e'.repeat(64)); }); test('the report accepts reused results only in the fast PR profile', () => { @@ -143,3 +147,73 @@ describe('E2E shard reuse through the runner', () => { expect(verifySliceResults(manifest, results).problems).toContain(`${FILE}: only the fast PR profile may reuse results; this lane executes fresh`); }); }); + +describe('planner-side panel reuse and negative receipts', () => { + const panelPlan = { kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined: false }; + const trialRequest = (trial: number, over: Partial = {}) => request({ + key: `${FILE}#setup-deploy-workflow~t${trial}`, panel: panelPlan, ...over }); + const source = (completedAt: number, runId = '1001/1') => ({ runId, revision: 'd'.repeat(40), completedAt }); + const panel = (key: string, outcomes: Array<'passed' | 'failed'>, completedAt = Date.now() - 1_000): PanelReceipt => ({ + schema: 1, key, case: 'setup-deploy-workflow', kind: 'behavior', panel: { n: 3, k: 2 }, source: source(completedAt), + trials: outcomes.map((outcome, i) => ({ trial: i + 1, outcome, ...(outcome === 'failed' ? { failure_class: 'timeout' as const } : {}) })), + }); + + test('every trial of a panel shares one identity; the panel policy is part of it', () => { + const key = (r: E2EShardReuseRequest) => { const x = e2eShardIdentity(r); if (x.status !== 'eligible') throw new Error(x.reason); return x.identity.key; }; + const t1 = key(trialRequest(1)); + expect(key(trialRequest(2, { env: laneEnv({ GSTACK_EVAL_TRIAL: '2' }) }))).toBe(t1); + expect(key(trialRequest(1, { panel: { ...panelPlan, quarantined: true } }))).not.toBe(t1); + expect(key(request())).not.toBe(t1); + }); + + test('a whole PASS panel receipt is reused per trial, a split PASS keeps its failed trial', () => { + const dir = path.join(scratch, 'panel-hit'); + const env = laneEnv({ EVALS_CACHE_DIR: dir }); + const reuse = prepareE2EShardReuse(trialRequest(2, { env }))!; + expect(reuse.lookupPanelTrial(2)).toBeNull(); + writePanelReceipt(dir, panel(reuse.inputKey, ['passed', 'failed', 'passed'])); + expect(reuse.lookupPanelTrial(2)).toMatchObject({ trial: { trial: 2, outcome: 'failed', failure_class: 'timeout' }, hit: { source: { runId: '1001/1' } } }); + expect(reuse.lookupPanelTrial(1)!.trial.outcome).toBe('passed'); + }); + + test('FAIL, partial, expired or negatively receipted panels are never reused', () => { + const dir = path.join(scratch, 'panel-miss'); + const key = 'a'.repeat(64); + for (const receipt of [panel(key, ['passed', 'failed', 'failed']), panel(key, ['passed', 'passed']), + panel(key, ['passed', 'passed', 'passed'], Date.now() - 2 * 24 * 60 * 60 * 1000)]) { + writePanelReceipt(dir, receipt); + expect(readPanelReceipt(dir, key)).toBeNull(); + } + writePanelReceipt(dir, panel(key, ['passed', 'passed', 'passed'], Date.now() - 5_000)); + expect(readPanelReceipt(dir, key)).not.toBeNull(); + writeNegativeReceipt(dir, { schema: 1, key, source: source(Date.now() - 1_000, '1002/1') }); + expect(readPanelReceipt(dir, key)).toBeNull(); + }); + + test('the planner ships one filtered set: a newer FAIL blocks an older PASS, an older FAIL does not', () => { + const from = path.join(scratch, 'select-from'); + const to = path.join(scratch, 'select-to'); + fs.mkdirSync(from, { recursive: true }); + const [blockedKey, keptKey, panelKey] = ['1', '2', '3'].map(c => c.repeat(64)); + const passReceipt = (key: string, completedAt: number) => fs.writeFileSync(path.join(from, `${key}.json`), + JSON.stringify({ schema: 1, proof: { source: source(completedAt) } })); + passReceipt(blockedKey, 1_000); + writeNegativeReceipt(from, { schema: 1, key: blockedKey, source: source(2_000, '1002/1') }); + passReceipt(keptKey, 3_000); + writeNegativeReceipt(from, { schema: 1, key: keptKey, source: source(2_000, '1002/1') }); + writePanelReceipt(from, panel(panelKey, ['passed', 'passed'])); + const result = selectPlanReceipts(from, to); + expect(result.blocked.sort()).toEqual([`${blockedKey}.json`, `${panelKey}.panel.json`].sort()); + expect(fs.readdirSync(to).sort()).toEqual([`${blockedKey}.fail.json`, `${keptKey}.fail.json`, `${keptKey}.json`].sort()); + }); + + test('merging receipt stores keeps the newest file per name', () => { + const [a, b, out] = ['merge-a', 'merge-b', 'merge-out'].map(name => path.join(scratch, name)); + const key = '4'.repeat(64); + writeNegativeReceipt(a, { schema: 1, key, source: source(5_000, '1/1') }); + writeNegativeReceipt(b, { schema: 1, key, source: source(9_000, '2/1') }); + expect(mergeReceiptDirs(out, [a, b, path.join(scratch, 'missing')])).toBe(2); + expect(JSON.parse(fs.readFileSync(path.join(out, `${key}.fail.json`), 'utf8')).source.runId).toBe('2/1'); + expect(mergeReceiptDirs(out, [a])).toBe(0); + }); +}); diff --git a/test/paid-report-fail-open.test.ts b/test/paid-report-fail-open.test.ts index 27343310c..86b9cc61e 100644 --- a/test/paid-report-fail-open.test.ts +++ b/test/paid-report-fail-open.test.ts @@ -31,7 +31,7 @@ function manifest(entries: PaidRunManifest['entries'], sliceCount: number): Paid } let caseCounter = 0; -function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record = {}) { +function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record = {}, env: NodeJS.ProcessEnv = {}) { const dir = path.join(base, `case-${++caseCounter}`); fs.mkdirSync(dir, { recursive: true }); fs.writeFileSync(path.join(dir, 'manifest.json'), JSON.stringify(plan)); @@ -41,7 +41,7 @@ function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record fs.writeFileSync(path.join(dir, name), JSON.stringify(body)); } const result = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--tier', plan.tier, '--report', dir], - { cwd: ROOT, encoding: 'utf8', timeout: 30_000, env: { ...process.env, EVALS_TIER: plan.tier } }); + { cwd: ROOT, encoding: 'utf8', timeout: 30_000, env: { ...process.env, GITHUB_RUN_ID: '', GITHUB_SHA: '', EVALS_TIER: plan.tier, ...env } }); return { status: result.status, out: `${result.stdout}\n${result.stderr}`, dir }; } @@ -219,4 +219,19 @@ describe('behavior and quarantined panels through --report', () => { expect(again.status).toBe(1); expect(again.stdout).toContain('attempt 1 (later attempts 2 reported, never replacing it)'); }); + + test('verdicts become next-run receipts: a whole fresh PASS panel, and negatives for FAIL panels', () => { + const inputKey = 'b'.repeat(64); + const withKey = (results: TrialResult[]) => [1, 2, 3, 4].map(index => slice(index, 4, index === 4 ? [passed(RULE_A)] + : [{ ...trialOutcome(index, results[index - 1]!)!, inputKey }])); + const env = { GITHUB_RUN_ID: '77', GITHUB_SHA: 'e'.repeat(40) }; + const green = report(plan(), withKey(['passed', 'failed', 'passed']), {}, env); + expect(green.status, green.out).toBe(0); + const receipt = JSON.parse(fs.readFileSync(path.join(green.dir, 'report-receipts', `${inputKey}.panel.json`), 'utf8')); + expect(receipt).toMatchObject({ key: inputKey, case: ID, panel: { n: 3, k: 2 }, source: { runId: '77/1' } }); + expect(receipt.trials.map((t: any) => t.outcome)).toEqual(['passed', 'failed', 'passed']); + const red = report(plan(), withKey(['passed', 'failed', 'failed']), {}, env); + expect(red.status).toBe(1); + expect(fs.readdirSync(path.join(red.dir, 'report-receipts'))).toEqual([`${inputKey}.fail.json`]); + }); }); diff --git a/test/paid-run-manifest.test.ts b/test/paid-run-manifest.test.ts index 74090aace..ddd9b96ce 100644 --- a/test/paid-run-manifest.test.ts +++ b/test/paid-run-manifest.test.ts @@ -39,6 +39,7 @@ import { verifySliceResults, expandTrialShards, formatCapacityPreflight, + panelReports, shardSlug, type PaidRunManifest, type ShardOutcome, @@ -607,4 +608,17 @@ describe('trial planner (behavior and quarantined panels)', () => { expect(packed.slices).toHaveLength(3); expect(Object.values(packed.estimates)).toEqual([180_000, 180_000, 180_000]); }); + + test('reuse is whole-panel only: a panel mixing reused and fresh trials is INCOMPLETE', () => { + const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' }); + const trials = manifest.entries.filter(entry => entry.trial); + const reused = { inputKey: 'c'.repeat(64), runId: '1001/1', revision: 'd'.repeat(40), completedAt: 1 }; + const results = (reusedTrials: number[]): SliceResult[] => trials.map(entry => ({ version: 1, tier: 'gate', sliceIndex: entry.slice, + sliceCount: manifest.sliceCount, outcomes: [{ files: [entry.file], status: 'passed', exitCode: 0, elapsedMs: 1, executedTests: 1, skippedTests: 0, + trial: { case: 'review-sql-injection', trial: Number(entry.file.slice(-1)), ...entry.trial!, outcome: 'passed', cost_usd: 0, duration_ms: 1 }, + ...(reusedTrials.includes(Number(entry.file.slice(-1))) ? { reused } : {}) }] })); + expect(panelReports(manifest, results([]), 1)[0]).toMatchObject({ status: 'PASS' }); + expect(panelReports(manifest, results([1, 2, 3]), 1)[0]).toMatchObject({ status: 'PASS' }); + expect(panelReports(manifest, results([2]), 1)[0]).toMatchObject({ status: 'INCOMPLETE', failsLane: true, reason: expect.stringContaining('partial panel reuse') }); + }); });