diff --git a/.github/workflows/evals.yml b/.github/workflows/evals.yml index 7cf43e744..8dbbc99bd 100644 --- a/.github/workflows/evals.yml +++ b/.github/workflows/evals.yml @@ -140,10 +140,23 @@ jobs: with: bun-version: 1.4.0 + # Planner-side reuse: restore this PR's newest receipt store (the report + # job saves one merged store per run) and ship ONE filtered set with the + # plan, so every trial of a panel sees the same receipts and a newer FAIL + # blocks any older PASS for the same inputs. + - name: Restore this PR's verified judge and E2E results + if: github.event_name == 'pull_request' + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 + with: + path: /tmp/gstack-eval-input-cache + key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-plan + restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}- + - name: Emit run manifest if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all' env: EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }} + EVALS_CACHE_DIR: ${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }} run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 16 - name: Emit validation-phase manifest @@ -182,7 +195,9 @@ jobs: - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 with: name: paid-plan - path: /tmp/paid-plan/manifest.json + path: | + /tmp/paid-plan/manifest.json + /tmp/paid-plan/receipts retention-days: 30 eval-slices: @@ -251,15 +266,13 @@ jobs: name: paid-plan path: /tmp/paid-plan - # Only this PR's receipts are eligible. No base-branch or cross-PR restore - # prefix; every receipt also verifies exact inputs and its original age. - - name: Restore this PR's verified judge and E2E results - if: github.event_name == 'pull_request' - uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 - with: - path: /tmp/gstack-eval-input-cache - key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }} - restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}- + # Receipts come only from the plan (this PR's store, filtered once by the + # planner); new receipts land beside the slice results and the report + # merges them into the next store. + - name: Seed this slice's receipts from the plan + run: | + mkdir -p /tmp/paid-slice-results/receipts + if [ -d /tmp/paid-plan/receipts ]; then cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/; fi - name: Run slice ${{ matrix.slice }} env: @@ -270,35 +283,12 @@ jobs: EVALS_JOBS: "2" EVALS_CONCURRENCY: "2" GSTACK_EVAL_DIR: /tmp/paid-slice-results - EVALS_CACHE_DIR: /tmp/gstack-eval-input-cache + EVALS_CACHE_DIR: /tmp/paid-slice-results/receipts EVALS_CACHE_REPOSITORY: ${{ github.repository }} EVALS_CACHE_PR: ${{ github.event.pull_request.number }} EVALS_CACHE_RUNTIME_ID: ${{ needs.build-image.outputs.runtime-id }} run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/paid-plan/manifest.json --slice ${{ matrix.slice }} - - name: Find finalized passing receipts - id: receipts - if: ${{ !cancelled() && github.event_name == 'pull_request' }} - run: | - # Only a producer publishes. A later reuse-only slice must not become - # the newest prefix match and hide another slice's newly earned pass. - for receipt in /tmp/gstack-eval-input-cache/*.json; do - [ -f "$receipt" ] || continue - if jq -e --arg run "$GITHUB_RUN_ID/$GITHUB_RUN_ATTEMPT" '.proof.source.runId == $run' "$receipt" >/dev/null 2>&1; then - echo 'present=true' >> "$GITHUB_OUTPUT" - break - fi - done - - # An unrelated failing case does not discard already verified passes. - # Failed/retried/partial attempts never become receipts in the first place. - - name: Save verified judge and E2E results for this PR - if: ${{ !cancelled() && steps.receipts.outputs.present == 'true' }} - uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 - with: - path: /tmp/gstack-eval-input-cache - key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }} - # Attempt-scoped: a re-run attempt's trials are reported under that # attempt and never replace (or collide with) the first attempt's. - name: Upload slice results @@ -396,8 +386,27 @@ jobs: echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT" - name: Stamp trial history series - if: always() && hashFiles('/tmp/paid-report/trial-outcomes.jsonl') != '' - run: bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl + if: always() + run: | + if [ -f /tmp/paid-report/trial-outcomes.jsonl ]; then + bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl + fi + + # One merged receipt store per run: the plan's shipped set, every slice's + # new pass receipts, and the report's panel and negative receipts. Saved + # last, so the next planner restores it as the newest prefix match. + - name: Merge this run's receipts + if: always() && github.event_name == 'pull_request' + run: | + bun --no-install run scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache \ + /tmp/paid-report/receipts /tmp/paid-report/report-receipts /tmp/paid-report/paid-slice-*/receipts + + - name: Save this PR's verified judge and E2E results + if: always() && github.event_name == 'pull_request' + uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 + with: + path: /tmp/gstack-eval-input-cache + key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged - name: Upload reconciliation output for the comment job if: always() @@ -501,7 +510,8 @@ jobs: COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.' STATUS="✅ PASS" - if [ "${RECONCILE_EXIT:-1}" != "0" ]; then STATUS="❌ FAIL"; fi + if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ] \ + || { [ -n "$VERIFIED" ] && [ "$(jq -r '.verdict.verdict' "$VERIFIED")" != "GREEN" ]; }; then STATUS="❌ FAIL"; fi if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (verified report unavailable)'; fi diff --git a/browse/test/dia-macos-qualification.test.ts b/browse/test/dia-macos-qualification.test.ts index 40e13bf67..48e483158 100644 --- a/browse/test/dia-macos-qualification.test.ts +++ b/browse/test/dia-macos-qualification.test.ts @@ -1214,7 +1214,7 @@ Binary Images: { status: 1, stdout: '', stderr: '' }, { status: 0, stdout: '', stderr: '' }, { status: 0, stdout: 'truncated-private-row', stderr: '' }, { status: null, stdout: null, stderr: null, error: new Error('synthetic-private-error') }, - ]) expect(inspectUidProcesses(23456, performance.now() + 10_000, {}, (() => result) as typeof spawnSync)).toEqual({ available: false }); + ]) expect(inspectUidProcesses(23456, performance.now() + 10_000, {}, (() => result) as unknown as typeof spawnSync)).toEqual({ available: false }); }); test('numeric UID process filtering runs through the real global process table', () => { @@ -1251,7 +1251,7 @@ Binary Images: { status: 113, stdout: '', stderr: 'Could not find domain for user uid: 23456' }, { status: null, stdout: null, stderr: null, error: new Error('synthetic-private-error') }, ]) { - const observation = inspectUserDomain(23456, performance.now() + 10_000, {}, (() => result) as typeof spawnSync); + const observation = inspectUserDomain(23456, performance.now() + 10_000, {}, (() => result) as unknown as typeof spawnSync); expect(observation.state).toBe('unavailable'); expect(observation.structure).toBeUndefined(); expect(JSON.stringify(observation)).not.toContain('synthetic-private'); diff --git a/scripts/e2e-shard-reuse.ts b/scripts/e2e-shard-reuse.ts index 692d89250..f0a905a65 100644 --- a/scripts/e2e-shard-reuse.ts +++ b/scripts/e2e-shard-reuse.ts @@ -30,6 +30,8 @@ import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from './eval-input-cache'; import { matchGlob } from '../test/helpers/test-selection'; import { E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from '../test/helpers/touchfiles-data'; +import { EVAL_CACHE_MAX_AGE_MS as RECEIPT_MAX_AGE_MS } from './eval-input-cache'; +import { panelVerdict, TRIAL_ENV, type EvalCaseKind, type PanelShape, type PanelTrial } from '../test/helpers/eval-store'; export interface E2EShardReuseRequest { root: string; @@ -50,6 +52,8 @@ export interface E2EShardReuseRequest { profile: string; /** The exact environment the child receives. */ env: NodeJS.ProcessEnv; + /** Isolated trial shard: its panel policy is part of the identity; the trial index is run-scoped. */ + panel?: { kind: EvalCaseKind; panel: PanelShape; quarantined: boolean }; } export interface E2EShardReuseHit { key: string; source: EvalPassingProof['source'] } @@ -61,7 +65,7 @@ const HARNESS_FILES = ['scripts/test-paid-shards.ts', 'scripts/e2e-shard-reuse.t const ENV_PREFIXES = ['EVALS_', 'GSTACK_', 'CLAUDE_', 'ANTHROPIC_', 'OPENAI_', 'GEMINI_', 'BUN_', 'NODE_', 'PLAYWRIGHT_']; /** Run-scoped values: provenance or transport, never behavior. Selection is bound as case ids. */ const RUN_SCOPED_ENV = new Set(['EVALS_RUN_ID', 'GSTACK_EVAL_DIR', 'EVALS_CACHE_DIR', 'EVALS_CACHE_PR', 'EVALS_CACHE_REPOSITORY', - 'EVALS_CACHE_RUNTIME_ID', 'EVALS_CACHE_PURPOSE', 'EVALS_SELECTION_JSON', 'EVALS_JUDGE_SELECTION_JSON']); + 'EVALS_CACHE_RUNTIME_ID', 'EVALS_CACHE_PURPOSE', 'EVALS_SELECTION_JSON', 'EVALS_JUDGE_SELECTION_JSON', TRIAL_ENV.trial]); const SECRET_ENV = /KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL/; /** The reuse-relevant environment the child sees; secrets contribute presence only. */ @@ -126,7 +130,8 @@ export function e2eShardIdentity(request: E2EShardReuseRequest): { status: 'elig coverage: { dependencies: 'complete', prompts: 'complete', environment: 'complete' }, unknownDependencies: [], files, prompts: Object.fromEntries(request.caseIds.map(id => [id, source])), - parameters: { rootPackage, key: request.key, caseIds: [...request.caseIds].sort(), casePattern: request.casePattern, + parameters: { rootPackage, key: request.panel ? request.key.replace(/~t\d+$/, '') : request.key, + ...(request.panel ? { panel: { kind: request.panel.kind, n: request.panel.panel.n, k: request.panel.panel.k, quarantined: request.panel.quarantined } } : {}), caseIds: [...request.caseIds].sort(), casePattern: request.casePattern, expectedCases: request.expectedCases, retries: request.retries, timeoutMs: request.timeoutMs, withinShardConcurrency: request.withinShardConcurrency, tier: request.tier, profile: request.profile, environment: e2eReuseEnvironment(env) }, @@ -156,7 +161,13 @@ const validResult = (identity: EvalInputIdentity, key: string) => (value: EvalCa * whose inputs did not change during execution. */ export function prepareE2EShardReuse(request: E2EShardReuseRequest): { + /** The input identity key: recorded on the outcome so the report can store verdicts against it. */ + inputKey: string; lookup(): E2EShardReuseHit | null; + /** Trial shards only: this trial's record from a whole PASS panel receipt of the plan's receipts. */ + lookupPanelTrial(trial: number): { hit: E2EShardReuseHit; trial: PanelTrial } | null; + /** True when the inputs are unchanged since `before` (the outcome may carry inputKey). */ + unchanged(): boolean; publish(): void; } | null { if (e2eReuseLaneProblem(request.env, 'pr') !== null) return null; @@ -164,6 +175,19 @@ export function prepareE2EShardReuse(request: E2EShardReuseRequest): { if (before.status !== 'eligible') return null; const common = { cacheDir: request.env.EVALS_CACHE_DIR!, purpose: 'gate' as const }; return { + inputKey: before.identity.key, + unchanged() { + const after = e2eShardIdentity(request); + return after.status === 'eligible' && after.identity.key === before.identity.key; + }, + lookupPanelTrial(trial) { + if (!request.panel) return null; + const receipt = readPanelReceipt(common.cacheDir, before.identity.key); + if (!receipt || receipt.case !== request.caseIds[0] || receipt.kind !== request.panel.kind + || receipt.panel.n !== request.panel.panel.n || receipt.panel.k !== request.panel.panel.k) return null; + const record = receipt.trials.find(t => t.trial === trial); + return record ? { hit: { key: receipt.key, source: receipt.source }, trial: record } : null; + }, lookup() { const found = lookupEvalInputCache({ ...common, identity: before.identity, validateResult: validResult(before.identity, request.key) }); return found.status === 'reused' ? { key: found.key, source: found.source } : null; @@ -184,3 +208,124 @@ export function prepareE2EShardReuse(request: E2EShardReuseRequest): { }, }; } + +// ─── Panel receipts, negative receipts and the planner's receipt selection ── +// +// Reuse is decided by the planner, once per panel: it ships the plan a +// receipt set in which every panel receipt is a whole PASS panel from one run +// and no pass receipt has a newer FAIL for the same identity. Executors look +// up only that set, so every trial of a panel sees the same receipts. The +// report writes panel receipts (all n trials fresh, one identity) and +// negative receipts (FAIL verdicts) after the verdict is known. + +export interface PanelReceipt { + schema: 1; + key: string; + case: string; + kind: EvalCaseKind; + panel: PanelShape; + trials: PanelTrial[]; + source: { runId: string; revision: string; completedAt: number }; +} + +export interface NegativeReceipt { schema: 1; key: string; source: { runId: string; revision: string; completedAt: number } } + +const RECEIPT_KEY = /^[a-f0-9]{64}$/; +const validSource = (source: any) => !!source && typeof source.runId === 'string' && /^[\w./-]{1,160}$/.test(source.runId) + && typeof source.revision === 'string' && /^[a-f0-9]{40}$/.test(source.revision) && Number.isSafeInteger(source.completedAt) && source.completedAt > 0; + +function readJson(file: string, maxBytes = 64 * 1024): any { + try { + const stat = fs.lstatSync(file); + if (!stat.isFile() || stat.size > maxBytes) return null; + return JSON.parse(fs.readFileSync(file, 'utf8')); + } catch { return null; } +} + +/** A whole, unexpired PASS panel receipt for `key`, re-verified with panelVerdict(); else null. */ +export function readPanelReceipt(cacheDir: string, key: string, now = Date.now()): PanelReceipt | null { + if (!RECEIPT_KEY.test(key)) return null; + const receipt = readJson(path.join(cacheDir, `${key}.panel.json`)); + if (!receipt || receipt.schema !== 1 || receipt.key !== key || typeof receipt.case !== 'string' || !validSource(receipt.source) + || receipt.source.completedAt > now || now - receipt.source.completedAt >= RECEIPT_MAX_AGE_MS || !Array.isArray(receipt.trials)) return null; + try { + const verdict = panelVerdict({ case: receipt.case, kind: receipt.kind, panel: receipt.panel, + trials: receipt.trials.map((t: PanelTrial) => ({ ...t, attempt: 1 })) }); + if (verdict.status !== 'PASS' || verdict.trials.length !== receipt.panel.n) return null; + } catch { return null; } + const negative = readJson(path.join(cacheDir, `${key}.fail.json`)); + if (negative && validSource(negative.source) && negative.source.completedAt >= receipt.source.completedAt) return null; + return receipt as PanelReceipt; +} + +export function writePanelReceipt(dir: string, receipt: PanelReceipt): void { + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, `${receipt.key}.panel.json`), `${JSON.stringify(receipt)}\n`, { mode: 0o600 }); +} + +export function writeNegativeReceipt(dir: string, receipt: NegativeReceipt): void { + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, `${receipt.key}.fail.json`), `${JSON.stringify(receipt)}\n`, { mode: 0o600 }); +} + +const receiptTime = (file: string): number => { + const parsed = readJson(file); + return Number(parsed?.source?.completedAt ?? parsed?.proof?.source?.completedAt) || 0; +}; + +/** + * Planner-side selection: copy `from` into `to`, dropping every pass or panel + * receipt that has a same-or-newer negative receipt for its identity, and + * every panel receipt that is not a whole PASS panel. Workflow-judge and + * other receipts pass through for their own validation at lookup. + */ +export function selectPlanReceipts(from: string, to: string, now = Date.now()): { shipped: number; blocked: string[] } { + fs.mkdirSync(to, { recursive: true }); + const blocked: string[] = []; + let shipped = 0; + let names: string[] = []; + try { names = fs.readdirSync(from).filter(name => name.endsWith('.json')); } catch { return { shipped, blocked }; } + for (const name of names) { + const file = path.join(from, name); + const [key, suffix] = [name.slice(0, 64), name.slice(64)]; + const negative = RECEIPT_KEY.test(key) ? readJson(path.join(from, `${key}.fail.json`)) : null; + const newerFail = negative && validSource(negative.source) && negative.source.completedAt >= receiptTime(file); + if (suffix === '.panel.json' && (newerFail || !readPanelReceipt(from, key, now))) { blocked.push(name); continue; } + if (suffix === '.json' && newerFail) { blocked.push(name); continue; } + fs.copyFileSync(file, path.join(to, name)); + shipped++; + } + return { shipped, blocked }; +} + +/** Merge receipt directories into one store, keeping the newest file per name. */ +export function mergeReceiptDirs(out: string, dirs: string[]): number { + fs.mkdirSync(out, { recursive: true }); + let merged = 0; + for (const dir of dirs) { + let names: string[] = []; + try { names = fs.readdirSync(dir).filter(name => name.endsWith('.json')); } catch { continue; } + for (const name of names) { + const source = path.join(dir, name); + const target = path.join(out, name); + if (!fs.lstatSync(source).isFile()) continue; + if (fs.existsSync(target) && receiptTime(target) >= receiptTime(source)) continue; + fs.copyFileSync(source, target); + merged++; + } + } + return merged; +} + +if (import.meta.main) { + const [command, first, ...rest] = process.argv.slice(2); + if (command === 'select' && first && rest[0]) { + const result = selectPlanReceipts(first, rest[0]); + console.log(`[e2e-reuse] shipped ${result.shipped} receipt(s) to the plan; blocked ${result.blocked.length} (newer FAIL or partial panel)`); + } else if (command === 'merge' && first) { + console.log(`[e2e-reuse] merged ${mergeReceiptDirs(first, rest)} receipt(s) into ${first}`); + } else { + console.error('usage: bun run scripts/e2e-shard-reuse.ts select | merge '); + process.exit(2); + } +} diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 198f8aceb..1a5d9c8b4 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -78,7 +78,7 @@ import { manualReviewProblem } from '../test/helpers/cookie-workflow-manual-revi import { preflightAnthropicApi } from '../test/helpers/anthropic-preflight'; import { OVERLAY_MIN_FILE_WALL_MS } from '../test/helpers/overlay-case-policy'; import { PR_PROFILE_CASE_IDS, PR_PROFILE_FILES, packageChangeOnlyVersion, selectPrProfile, type PrProfileSelection } from './test-pr-profile'; -import { e2eReuseLaneProblem, prepareE2EShardReuse } from './e2e-shard-reuse'; +import { e2eReuseLaneProblem, prepareE2EShardReuse, selectPlanReceipts, writeNegativeReceipt, writePanelReceipt } from './e2e-shard-reuse'; type E2EShardReuse = NonNullable>; import { @@ -826,6 +826,8 @@ export interface ShardOutcome { reused?: { inputKey: string; runId: string; revision: string; completedAt: number }; /** The parent could not run the shard at all (a runner error, never a trial verdict). */ runnerError?: string; + /** PR lane: the reuse input identity of a freshly executed shard whose inputs stayed unchanged. */ + inputKey?: string; /** Isolated trial shards only: the trial record this shard produced. */ trial?: ShardTrialRecord; } @@ -1057,7 +1059,22 @@ export async function runPaidShard( // Bootstrap-retention qualification binds per-run state, so that shard stays fresh. const reuse = files.some(file => normalizeRelativePath(file) === 'test/skill-e2e-qa-workflow.test.ts') ? null : options.reuseFor?.(files, env, budget) ?? null; - const reused = reuse?.lookup() ?? null; + // A trial reuses only its record from a whole PASS panel receipt the + // planner shipped; a single trial never has a pass receipt of its own. + const panelHit = trialPlan && trialIndex !== null ? reuse?.lookupPanelTrial(trialIndex) ?? null : null; + if (panelHit && trialPlan && trialIndex !== null && caseId !== null) { + const reusedFrom = { inputKey: panelHit.hit.key, runId: panelHit.hit.source.runId, revision: panelHit.hit.source.revision, completedAt: panelHit.hit.source.completedAt }; + const passedTrial = panelHit.trial.outcome === 'passed'; + log(`${label} REUSED trial ${trialIndex}/${trialPlan.panel.n} of ${caseId} (${panelHit.trial.outcome}) from the whole PASS panel of run ${reusedFrom.runId}`); + return { shard: shardNumber, files, status: passedTrial ? 'passed' : 'failed', exitCode: passedTrial ? 0 : 1, elapsedMs: 0, groupPid: null, + executedTests: 1, skippedTests: 0, budget, reused: reusedFrom, + trial: { case: caseId, trial: trialIndex, kind: trialPlan.kind, panel: trialPlan.panel, quarantined: trialPlan.quarantined, + outcome: panelHit.trial.outcome, cost_usd: 0, duration_ms: 0, + ...(panelHit.trial.failure_class ? { failure_class: panelHit.trial.failure_class } : {}), + ...(panelHit.trial.exit_reason ? { exit_reason: panelHit.trial.exit_reason } : {}), + ...(panelHit.trial.error ? { error: panelHit.trial.error } : {}) } }; + } + const reused = trialPlan ? null : reuse?.lookup() ?? null; if (reused) { const reusedFrom = { input_key: reused.key, run_id: reused.source.runId, revision: reused.source.revision, completed_at: new Date(reused.source.completedAt).toISOString() }; @@ -1255,7 +1272,8 @@ export async function runPaidShard( } } const elapsedMs = Date.now() - startedAt; - if (status === 'passed' && reuse) reuse.publish(); + if (status === 'passed' && reuse && !trialPlan) reuse.publish(); + const inputKey = reuse?.unchanged() ? reuse.inputKey : undefined; // Failure debuggability without the RAM cost: read back only the log's // tail. Live mode already streamed everything, so no re-print there. @@ -1273,7 +1291,8 @@ export async function runPaidShard( ? summary.terminalTestCounts.reduce((a, b) => a + b, 0) : null; const skippedTests = summary.terminalTestCounts.length > 0 ? summary.skippedTests : null; - return withTrial({ shard: shardNumber, files, status, exitCode, elapsedMs, groupPid, executedTests, skippedTests, budget }); + return withTrial({ shard: shardNumber, files, status, exitCode, elapsedMs, groupPid, executedTests, skippedTests, budget, + ...(inputKey ? { inputKey } : {}) }); } export interface RunSummary { @@ -2027,7 +2046,7 @@ export interface SliceResult { /** Epoch ms bounds of the slice's shard execution (lane wall time). */ startedAt?: number; finishedAt?: number; - outcomes: Array>; + outcomes: Array>; } /** @@ -2113,11 +2132,13 @@ export function verifySliceResults( if (outcome.reused !== undefined) { const r = outcome.reused; if (manifest.prCoverage?.mode !== 'pr') problems.push(`${file}: only the fast PR profile may reuse results; this lane executes fresh`); - if (outcome.status !== 'passed' || outcome.exitCode !== 0 || !/^[a-f0-9]{64}$/.test(r?.inputKey ?? '') + const reusedVerdictOk = outcome.trial ? outcome.trial.outcome !== null : outcome.status === 'passed' && outcome.exitCode === 0; + if (!reusedVerdictOk || !/^[a-f0-9]{64}$/.test(r?.inputKey ?? '') || !/^[\w./-]{1,160}$/.test(r?.runId ?? '') || !/^[a-f0-9]{40}$/.test(r?.revision ?? '') || !Number.isSafeInteger(r?.completedAt) || r.completedAt <= 0) { problems.push(`${file}: malformed reused result`); } } + if (outcome.inputKey !== undefined && !/^[a-f0-9]{64}$/.test(outcome.inputKey)) problems.push(`${file}: malformed input identity`); if (reported.has(file)) problems.push(`${file} reported by two slices`); reported.set(file, { slice: result.sliceIndex, status: outcome.status, ...(outcome.trial ? { trial: outcome.trial } : {}) }); const registered = FILE_RETRY_BUDGETS.find(budget => budget.file === shardFile(file)); @@ -2301,6 +2322,12 @@ export function panelReports(manifest: PaidRunManifest, results: SliceResult[], ...(t.timeout_at_turn !== undefined ? { timeout_at_turn: t.timeout_at_turn } : {}) }]; }); const verdict = panelVerdict({ case: shardCaseId(key)!, kind: plan.kind, panel: plan.panel, trials, quarantined: plan.quarantined }); + // Reuse is whole-panel only: every trial reused from one run, or none. + const sources = entries.map(entry => reported.get(normalizeRelativePath(entry.file))?.outcome.reused?.runId ?? null); + if (sources.some(source => source !== null) && (sources.some(source => source === null) || new Set(sources).size > 1)) { + return { ...verdict, status: 'INCOMPLETE', split: false, failsLane: true, coverage: false, redClass: 'INCOMPLETE', + reason: 'partial panel reuse (every trial must come from one reused panel, or none)', file: shardFile(key), slices }; + } return { ...verdict, file: shardFile(key), slices }; }); } @@ -2507,6 +2534,29 @@ export function runPaidReport(reportDir: string, options: { writeDurations?: boo const laterPanels = attempts.slice(1).flatMap(attempt => panelReports(manifest, artifacts.map(a => a.result), attempt) .filter(panel => panel.trials.length > 0)); + // PR-lane receipts from verdicts (the planner ships them to the next run): + // a whole fresh PASS panel with one input identity becomes a panel receipt; + // a FAIL panel or a failed rule shard becomes a negative receipt that + // blocks reuse of any older PASS for the same identity. + const revision = env.GITHUB_SHA ?? ''; + if (env.GITHUB_RUN_ID && /^[a-f0-9]{40}$/.test(revision)) { + const receiptsDir = path.join(reportDir, 'report-receipts'); + const source = { runId: `${env.GITHUB_RUN_ID}/${primary}`, revision, completedAt: Date.now() }; + const outcomes = results.flatMap(r => r.outcomes); + for (const panel of panels) { + const keys = outcomes.filter(o => o.trial?.case === panel.case && shardFile(o.files[0] ?? '') === panel.file).map(o => o.reused ? null : o.inputKey ?? null); + if (keys.length !== panel.panel.n || keys.some(k => k === null) || new Set(keys).size !== 1) continue; + if (panel.status === 'PASS') { + writePanelReceipt(receiptsDir, { schema: 1, key: keys[0]!, case: panel.case, kind: panel.kind, panel: panel.panel, source, + trials: panel.trials.map(({ trial, outcome, failure_class, exit_reason, error }) => ({ trial, outcome, + ...(failure_class ? { failure_class } : {}), ...(exit_reason ? { exit_reason } : {}), ...(error ? { error } : {}) })) }); + } else if (panel.status === 'FAIL') writeNegativeReceipt(receiptsDir, { schema: 1, key: keys[0]!, source }); + } + for (const outcome of outcomes.filter(o => !o.trial && o.inputKey && !o.reused && o.status !== 'passed')) { + writeNegativeReceipt(receiptsDir, { schema: 1, key: outcome.inputKey!, source }); + } + } + // Quarantine cap and expiry are the weekly pass-rates gate's (eval-flake-rank --gate). // History: one trial-outcomes line per isolated trial and per JUnit rule/judge case. @@ -2796,6 +2846,12 @@ async function main(): Promise { ); for (const line of formatSlicePlan(manifest)) console.log(line); for (const line of formatCapacityPreflight(manifest, options.maxParallel ?? undefined)) console.log(line); + // Planner-side reuse: ship ONE filtered receipt set with the plan, so + // every trial of a panel (on any slice) sees the same receipts. + if (manifest.profile === 'pr' && process.env.EVALS_CACHE_DIR) { + const shipped = selectPlanReceipts(process.env.EVALS_CACHE_DIR, path.join(path.dirname(path.resolve(options.emitPlanPath)), 'receipts')); + console.log(`[test:paid] reuse: shipped ${shipped.shipped} receipt(s) with the plan; blocked ${shipped.blocked.length} (newer FAIL or partial panel)`); + } return 0; } @@ -2865,6 +2921,7 @@ async function main(): Promise { const { registered, known } = fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')); const exclude = mine.find(entry => entry.file === key)?.excludeCases; return prepareE2EShardReuse({ root: ROOT, key, file, caseIds: prProfileShardIds(key, manifest.selection!, exclude), + ...(trials[normalizeRelativePath(key)] ? { panel: trials[normalizeRelativePath(key)] } : {}), registeredIds: registered, registrationKnown: known, casePattern: prProfileTestNamePattern(key, manifest.selection!, exclude), expectedCases: expectedPrCaseCount(key, manifest.selection!, exclude), retries: retriesForFiles(files), timeoutMs: budget.timeoutMs, withinShardConcurrency: options.withinShardConcurrency, @@ -2901,9 +2958,9 @@ async function main(): Promise { attempt: Number.isSafeInteger(attempt) && attempt > 0 ? attempt : 1, startedAt, finishedAt: Date.now(), - outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused, runnerError, trial }) => + outcomes: guarded.map(({ files, status, exitCode, elapsedMs, executedTests, skippedTests, budget, reused, runnerError, trial, inputKey }) => ({ files, status, exitCode, elapsedMs, executedTests, skippedTests, ...(budget ? { budget } : {}), ...(reused ? { reused } : {}), - ...(runnerError !== undefined ? { runnerError } : {}), ...(trial ? { trial } : {}) })), + ...(runnerError !== undefined ? { runnerError } : {}), ...(trial ? { trial } : {}), ...(inputKey ? { inputKey } : {}) })), }; fs.mkdirSync(evalDirBase, { recursive: true }); const sliceResultPath = path.join(evalDirBase, `slice-${options.sliceIndex}.json`); diff --git a/scripts/typecheck-test-baseline.json b/scripts/typecheck-test-baseline.json index 612942cf7..c96077b9e 100644 --- a/scripts/typecheck-test-baseline.json +++ b/scripts/typecheck-test-baseline.json @@ -36,7 +36,6 @@ "browse/test/cookie-import-transport.test.ts\tTS2741\tProperty 'preconnect' is missing in type '() => Promise' but required in type 'typeof fetch'.": 1, "browse/test/dia-gui-readiness.test.ts\tTS2352\tConversion of type '() => { status: number; stdout: string; stderr: string; }' to type '{ (command: string): SpawnSyncReturns; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ status: number; stdout: string; stderr: string; }' is missing the following properties from type 'SpawnSyncReturns': pid, output, signal": 1, "browse/test/dia-launch-comparison.test.ts\tTS7016\tCould not find a declaration file for module '../../.github/scripts/dia-launch-driver.mjs'. '.github/scripts/dia-launch-driver.mjs' implicitly has an 'any' type.": 1, - "browse/test/dia-macos-qualification.test.ts\tTS2352\tConversion of type '() => { error?: undefined; status: number; stdout: string; stderr: string; } | { status: null; stdout: null; stderr: null; error: Error; }' to type '{ (command: string): SpawnSyncReturns; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ error?: undefined; status: number; stdout: string; stderr: string; } | { status: null; stdout: null; stderr: null; error: Error; }' is not comparable to type 'SpawnSyncReturns'. Type '{ status: null; stdout: null; stderr: null; error: Error; }' is missing the following properties from type 'SpawnSyncReturns': pid, output, signal": 2, "browse/test/dia-macos-qualification.test.ts\tTS2352\tConversion of type '() => { status: number; stdout: string; stderr: string; }' to type '{ (command: string): SpawnSyncReturns; (command: string, options: SpawnSyncOptionsWithStringEncoding): SpawnSyncReturns<...>; (command: string, options: SpawnSyncOptionsWithBufferEncoding): SpawnSyncReturns<...>; (command: string, options?: SpawnSyncOptions | undefined): SpawnSyncReturns<...>; (comm...' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ status: number; stdout: string; stderr: string; }' is missing the following properties from type 'SpawnSyncReturns': pid, output, signal": 2, "browse/test/dia-macos-qualification.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type 'string' is not assignable to parameter of type '\"browser_profile_unavailable\" | \"code_signing_error\" | \"debugging_pipe_unavailable\" | \"default_profile_policy\" | \"dynamic_library_error\" | \"graphics_or_bootstrap_error\" | \"keychain_access_failed\" | \"keychain_interaction_disallowed\" | \"keychain_interaction_required\"'.": 1, "browse/test/domain-skills-e2e.test.ts\tTS2339\tProperty 'cleanup' does not exist on type 'BrowserManager'.": 1, diff --git a/test/ci-eval-cache.test.ts b/test/ci-eval-cache.test.ts index a34f430f5..ce2fe5618 100644 --- a/test/ci-eval-cache.test.ts +++ b/test/ci-eval-cache.test.ts @@ -21,19 +21,31 @@ test('only PR runs select the fast profile; manual and scheduled coverage stays } }); -test('receipt transport restores only this repository and PR with no broad fallback key', () => { - const steps = paid.jobs['eval-slices'].steps; - const restore = steps.filter((s: any) => s.uses?.startsWith('actions/cache/restore@')); - const save = steps.filter((s: any) => s.uses?.startsWith('actions/cache/save@')); +test('receipt transport: the planner restores only this repository and PR, the report saves one merged store', () => { + const planner = paid.jobs['plan-slices'].steps; + const restore = planner.filter((s: any) => s.uses?.startsWith('actions/cache/restore@')); expect(restore).toHaveLength(1); - expect(save).toHaveLength(1); expect(restore[0].if).toBe("github.event_name == 'pull_request'"); + expect(restore[0].with.path).toBe('/tmp/gstack-eval-input-cache'); expect(restore[0].with['restore-keys']).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-'); - expect(save[0].with.key).toBe(restore[0].with.key); - expect(save[0].with.key).toContain('${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}'); + const emit = planner.find((s: any) => s.run?.includes('--emit-plan /tmp/paid-plan/manifest.json')); + expect(emit.env.EVALS_CACHE_DIR).toBe("${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }}"); + const upload = planner.find((s: any) => s.with?.name === 'paid-plan'); + expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/paid-plan/manifest.json', '/tmp/paid-plan/receipts']); + // Executors never restore or save a cache of their own: every slice sees the plan's one receipt set. + const executor = paid.jobs['eval-slices'].steps; + expect(executor.filter((s: any) => s.uses?.startsWith('actions/cache/'))).toHaveLength(0); + expect(executor.find((s: any) => s.name === "Seed this slice's receipts from the plan").run).toContain('cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/'); + const report = paid.jobs['slices-report'].steps; + const merge = report.find((s: any) => s.name === "Merge this run's receipts"); + expect(merge.run).toContain('scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache'); + const save = report.filter((s: any) => s.uses?.startsWith('actions/cache/save@')); + expect(save).toHaveLength(1); expect(save[0].with.path).toBe('/tmp/gstack-eval-input-cache'); - expect(save[0].if).toContain("steps.receipts.outputs.present == 'true'"); + expect(save[0].with.key).toBe('eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged'); + expect(report.indexOf(save[0])).toBeGreaterThan(report.indexOf(merge)); expect(paid.jobs['eval-slices'].permissions).toEqual({ contents: 'read', packages: 'read' }); + expect(paid.jobs['slices-report'].permissions).toEqual({ contents: 'read' }); expect(JSON.stringify(periodic)).not.toContain('actions/cache/'); }); @@ -43,56 +55,21 @@ test('the judge binds cache receipts to the PR and installed runtime, not the co expect(runtime.run).toContain('sha256sum /tmp/eval-runtime-manifest.json'); const run = paid.jobs['eval-slices'].steps.find((s: any) => s.run?.includes('--plan /tmp/paid-plan/manifest.json')); expect(run.env).toMatchObject({ - EVALS_CACHE_DIR: '/tmp/gstack-eval-input-cache', + EVALS_CACHE_DIR: '/tmp/paid-slice-results/receipts', EVALS_CACHE_REPOSITORY: '${{ github.repository }}', EVALS_CACHE_PR: '${{ github.event.pull_request.number }}', EVALS_CACHE_RUNTIME_ID: '${{ needs.build-image.outputs.runtime-id }}', }); }); -test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('only a new passing producer can publish the next cache snapshot', () => { - const directory = mkdtempSync(join(tmpdir(), 'ci-cache-producer-')); - const receipts = join(directory, 'receipts'); - const output = join(directory, 'output'); - mkdirSync(receipts); - const step = paid.jobs['eval-slices'].steps.find((s: any) => s.id === 'receipts'); - const script = step.run.replaceAll('/tmp/gstack-eval-input-cache', receipts); - const run = () => { - writeFileSync(output, ''); - const result = spawnSync('bash', ['-e', '-c', script], { - env: { ...process.env, GITHUB_OUTPUT: output, GITHUB_RUN_ID: '42', GITHUB_RUN_ATTEMPT: '2' }, - encoding: 'utf8', timeout: 5000, - }); - expect(result.status, result.stderr).toBe(0); - return readFileSync(output, 'utf8'); - }; - try { - expect(run()).toBe(''); - writeFileSync(join(receipts, 'old.json'), JSON.stringify({ proof: { source: { runId: '41/1' } } })); - writeFileSync(join(receipts, 'corrupt.json'), '{'); - expect(run()).toBe(''); - writeFileSync(join(receipts, 'prior-attempt.json'), JSON.stringify({ proof: { source: { runId: '42/1' } } })); - expect(run()).toBe(''); - writeFileSync(join(receipts, 'fresh.json'), JSON.stringify({ proof: { source: { runId: '42/2' } } })); - expect(run()).toBe('present=true\n'); - } finally { rmSync(directory, { recursive: true, force: true }); } -}); - -test.skipIf(!Bun.which('jq'))('the actual comment separates reused evidence, retry outcomes and deferred coverage', () => { +test.skipIf(!Bun.which('jq'))('the actual comment shows deferred coverage and never recomputes a verdict', () => { const comment = paid.jobs['slices-comment'].steps.find((s: any) => s.name === 'Post PR comment').run as string; const evaluate = (filter: string, value: unknown) => { const result = spawnSync('jq', ['-r', filter], { input: JSON.stringify(value), encoding: 'utf8', timeout: 5000 }); expect(result.status, result.stderr).toBe(0); return result.stdout.trim(); }; - const stats = comment.match(/STATS=\$\(jq -r '([^']+)'/)![1]!; - expect(evaluate(stats, { tests: [ - { name: 'retry', passed: false }, { name: 'retry', passed: true }, - { name: 'exhausted', passed: false }, { name: 'exhausted', passed: false }, - { name: 'regressed', passed: true }, { name: 'regressed', passed: false }, - { name: 'reused', passed: true, execution: 'reused' }, - ], flaky_retries: ['retry', 'exhausted', 'regressed'].map(name => ({ name, attempts: 2 })) })).toBe('4 2 2 3 3 1'); - expect(comment).toContain("printf ' | ⚠ %s cases with multiple attempts'"); + expect(comment).not.toContain('group_by(.name)'); expect(comment).not.toMatch(/flaky pass\(es\)|passed only on retry|not blocking/); const coverage = comment.match(/COVERAGE=\$\(jq -r '([^']+)'/)![1]!; const text = evaluate(coverage, { profile: 'pr', selection: { e2e: ['probe'], judges: ['judge'] }, @@ -107,9 +84,9 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f const job = paid.jobs['slices-comment']; expect(job.permissions).toMatchObject({ 'pull-requests': 'write' }); expect(JSON.stringify(job.steps)).not.toMatch(/actions\/checkout|setup-bun|bun run|npm |node /); - const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict'); - expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json']); - expect(job.steps.find((step: any) => step.with?.name === 'report-verdict').with.path).toBe('/tmp/verdict'); + const upload = paid.jobs['slices-report'].steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}'); + expect(upload.with.path.trim().split('\n')).toEqual(['/tmp/report.txt', '/tmp/paid-report/collector-outcomes.json', '/tmp/paid-report/report-summary.md']); + expect(job.steps.find((step: any) => step.with?.name === 'report-verdict-a${{ github.run_attempt }}').with.path).toBe('/tmp/verdict'); const root = mkdtempSync(join(tmpdir(), 'ci-comment-')); const paidDir = join(root, 'paid-report'); const verdictDir = join(root, 'verdict'); @@ -123,9 +100,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f writeFileSync(join(paidDir, 'judge.json'), JSON.stringify({ total_tests: 2, tier: 'llm-judge', shard: 1, tests: [{ name: 'manual', passed: false, manual_review: { unverified: true } }, { name: 'reused', passed: true, execution: 'reused' }], flaky_retries: [] })); - const summary = { version: 1, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0, + const summary = { version: 2, files: [{ file: 'judge.json', tier: 'llm-judge', shard: 1, cost: 0, total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 }], - totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 } }; + totals: { total: 2, passed: 1, failed: 0, manual_accepted: 1, executed: 1, reused: 1, attempts: 2, flaky: 0 }, + verdict: { verdict: 'GREEN' }, headline: ['[test:paid] VERDICT GREEN — lane gate/pr, attempt 1'], panels: [], + failures: ['⚠ case-x behavior PASS 2/3 (✓✗✓) t2: timeout at turn 3 — @\u200bsomeone said no'] }; mkdirSync(join(verdictDir, 'paid-report')); const summaryPath = join(verdictDir, 'paid-report/collector-outcomes.json'); const script = (job.steps.find((step: any) => step.name === 'Post PR comment').run as string) @@ -142,8 +121,11 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f const verified = run(); expect(verified.status, verified.stderr).toBe(0); expect(verified.stdout).toContain('⚠ MANUAL ACCEPTED (unscored)'); - expect(verified.stdout).toContain('1 automated passed / 2 final results'); - expect(verified.stdout).toContain('0 failed, 1 manual accepted'); + expect(verified.stdout).toContain('VERDICT GREEN — lane gate/pr, attempt 1'); + expect(verified.stdout).toContain('1 executed, 1 reused** rule/judge records'); + expect(verified.stdout).toContain('1 manual accepted'); + expect(verified.stdout).toContain('### Failures and split verdicts'); + expect(verified.stdout).toContain('PASS 2/3 (✓✗✓) t2: timeout at turn 3'); const unrelatedFailure = { ...summary, files: [{ ...summary.files[0], total: 3, failed: 1, executed: 2, attempts: 3 }], totals: { ...summary.totals, total: 3, failed: 1, @@ -152,16 +134,20 @@ test.skipIf(!Bun.which('jq') || !Bun.which('bash'))('comment consumes verified f const red = run(); expect(red.status, red.stderr).toBe(0); expect(red.stdout).toContain('❌ FAIL'); - expect(red.stdout).toContain('1 failed, 1 manual accepted'); + + writeFileSync(summaryPath, JSON.stringify({ ...summary, verdict: { verdict: 'RED' } })); + const redVerdict = run(); + expect(redVerdict.status, redVerdict.stderr).toBe(0); + expect(redVerdict.stdout).toContain('❌ FAIL'); writeFileSync(summaryPath, JSON.stringify({ ...summary, totals: { ...summary.totals, manual_accepted: 2 } })); const tampered = run(); expect(tampered.status, tampered.stderr).toBe(0); - expect(tampered.stdout).toContain('manual acceptance unavailable/unverified'); + expect(tampered.stdout).toContain('verified report unavailable'); expect(tampered.stdout).not.toContain('⚠ MANUAL ACCEPTED (unscored)'); rmSync(summaryPath); const absent = run(); expect(absent.status, absent.stderr).toBe(0); - expect(absent.stdout).toContain('manual acceptance unavailable/unverified'); + expect(absent.stdout).toContain('verified report unavailable'); } finally { rmSync(root, { recursive: true, force: true }); } }); diff --git a/test/e2e-shard-reuse.test.ts b/test/e2e-shard-reuse.test.ts index a07b8f8e5..acb515321 100644 --- a/test/e2e-shard-reuse.test.ts +++ b/test/e2e-shard-reuse.test.ts @@ -4,7 +4,8 @@ import * as os from 'node:os'; import * as path from 'node:path'; import { e2eReuseEnvironment, e2eReuseLaneProblem, e2eShardIdentity, e2eShardInputFiles, prepareE2EShardReuse, - type E2EShardReuseRequest, + mergeReceiptDirs, readPanelReceipt, selectPlanReceipts, writeNegativeReceipt, writePanelReceipt, + type E2EShardReuseRequest, type PanelReceipt, } from '../scripts/e2e-shard-reuse'; import { buildRunManifest, fileCaseRegistration, runPaidShard, verifySliceResults, type SliceResult } from '../scripts/test-paid-shards'; @@ -127,10 +128,13 @@ describe('E2E shard reuse through the runner', () => { test('a failed shard never publishes a receipt', async () => { let published = 0; const outcome = await runPaidShard([FILE], 1, 1, { rootDir: ROOT, logDir: scratch, env: laneEnv(), log: () => {}, - reuseFor: () => ({ lookup: () => null, publish: () => { published++; } }), + reuseFor: () => ({ inputKey: 'e'.repeat(64), unchanged: () => true, lookupPanelTrial: () => null, + lookup: () => null, publish: () => { published++; } }), commandFor: () => ({ command: process.execPath, args: ['-e', 'process.exit(1)'] }) }); expect(outcome.status).toBe('failed'); expect(published).toBe(0); + // The identity rides on the outcome so the report can store the FAIL as a negative receipt. + expect(outcome.inputKey).toBe('e'.repeat(64)); }); test('the report accepts reused results only in the fast PR profile', () => { @@ -143,3 +147,73 @@ describe('E2E shard reuse through the runner', () => { expect(verifySliceResults(manifest, results).problems).toContain(`${FILE}: only the fast PR profile may reuse results; this lane executes fresh`); }); }); + +describe('planner-side panel reuse and negative receipts', () => { + const panelPlan = { kind: 'behavior' as const, panel: { n: 3, k: 2 }, quarantined: false }; + const trialRequest = (trial: number, over: Partial = {}) => request({ + key: `${FILE}#setup-deploy-workflow~t${trial}`, panel: panelPlan, ...over }); + const source = (completedAt: number, runId = '1001/1') => ({ runId, revision: 'd'.repeat(40), completedAt }); + const panel = (key: string, outcomes: Array<'passed' | 'failed'>, completedAt = Date.now() - 1_000): PanelReceipt => ({ + schema: 1, key, case: 'setup-deploy-workflow', kind: 'behavior', panel: { n: 3, k: 2 }, source: source(completedAt), + trials: outcomes.map((outcome, i) => ({ trial: i + 1, outcome, ...(outcome === 'failed' ? { failure_class: 'timeout' as const } : {}) })), + }); + + test('every trial of a panel shares one identity; the panel policy is part of it', () => { + const key = (r: E2EShardReuseRequest) => { const x = e2eShardIdentity(r); if (x.status !== 'eligible') throw new Error(x.reason); return x.identity.key; }; + const t1 = key(trialRequest(1)); + expect(key(trialRequest(2, { env: laneEnv({ GSTACK_EVAL_TRIAL: '2' }) }))).toBe(t1); + expect(key(trialRequest(1, { panel: { ...panelPlan, quarantined: true } }))).not.toBe(t1); + expect(key(request())).not.toBe(t1); + }); + + test('a whole PASS panel receipt is reused per trial, a split PASS keeps its failed trial', () => { + const dir = path.join(scratch, 'panel-hit'); + const env = laneEnv({ EVALS_CACHE_DIR: dir }); + const reuse = prepareE2EShardReuse(trialRequest(2, { env }))!; + expect(reuse.lookupPanelTrial(2)).toBeNull(); + writePanelReceipt(dir, panel(reuse.inputKey, ['passed', 'failed', 'passed'])); + expect(reuse.lookupPanelTrial(2)).toMatchObject({ trial: { trial: 2, outcome: 'failed', failure_class: 'timeout' }, hit: { source: { runId: '1001/1' } } }); + expect(reuse.lookupPanelTrial(1)!.trial.outcome).toBe('passed'); + }); + + test('FAIL, partial, expired or negatively receipted panels are never reused', () => { + const dir = path.join(scratch, 'panel-miss'); + const key = 'a'.repeat(64); + for (const receipt of [panel(key, ['passed', 'failed', 'failed']), panel(key, ['passed', 'passed']), + panel(key, ['passed', 'passed', 'passed'], Date.now() - 2 * 24 * 60 * 60 * 1000)]) { + writePanelReceipt(dir, receipt); + expect(readPanelReceipt(dir, key)).toBeNull(); + } + writePanelReceipt(dir, panel(key, ['passed', 'passed', 'passed'], Date.now() - 5_000)); + expect(readPanelReceipt(dir, key)).not.toBeNull(); + writeNegativeReceipt(dir, { schema: 1, key, source: source(Date.now() - 1_000, '1002/1') }); + expect(readPanelReceipt(dir, key)).toBeNull(); + }); + + test('the planner ships one filtered set: a newer FAIL blocks an older PASS, an older FAIL does not', () => { + const from = path.join(scratch, 'select-from'); + const to = path.join(scratch, 'select-to'); + fs.mkdirSync(from, { recursive: true }); + const [blockedKey, keptKey, panelKey] = ['1', '2', '3'].map(c => c.repeat(64)); + const passReceipt = (key: string, completedAt: number) => fs.writeFileSync(path.join(from, `${key}.json`), + JSON.stringify({ schema: 1, proof: { source: source(completedAt) } })); + passReceipt(blockedKey, 1_000); + writeNegativeReceipt(from, { schema: 1, key: blockedKey, source: source(2_000, '1002/1') }); + passReceipt(keptKey, 3_000); + writeNegativeReceipt(from, { schema: 1, key: keptKey, source: source(2_000, '1002/1') }); + writePanelReceipt(from, panel(panelKey, ['passed', 'passed'])); + const result = selectPlanReceipts(from, to); + expect(result.blocked.sort()).toEqual([`${blockedKey}.json`, `${panelKey}.panel.json`].sort()); + expect(fs.readdirSync(to).sort()).toEqual([`${blockedKey}.fail.json`, `${keptKey}.fail.json`, `${keptKey}.json`].sort()); + }); + + test('merging receipt stores keeps the newest file per name', () => { + const [a, b, out] = ['merge-a', 'merge-b', 'merge-out'].map(name => path.join(scratch, name)); + const key = '4'.repeat(64); + writeNegativeReceipt(a, { schema: 1, key, source: source(5_000, '1/1') }); + writeNegativeReceipt(b, { schema: 1, key, source: source(9_000, '2/1') }); + expect(mergeReceiptDirs(out, [a, b, path.join(scratch, 'missing')])).toBe(2); + expect(JSON.parse(fs.readFileSync(path.join(out, `${key}.fail.json`), 'utf8')).source.runId).toBe('2/1'); + expect(mergeReceiptDirs(out, [a])).toBe(0); + }); +}); diff --git a/test/paid-report-fail-open.test.ts b/test/paid-report-fail-open.test.ts index 27343310c..86b9cc61e 100644 --- a/test/paid-report-fail-open.test.ts +++ b/test/paid-report-fail-open.test.ts @@ -31,7 +31,7 @@ function manifest(entries: PaidRunManifest['entries'], sliceCount: number): Paid } let caseCounter = 0; -function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record = {}) { +function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record = {}, env: NodeJS.ProcessEnv = {}) { const dir = path.join(base, `case-${++caseCounter}`); fs.mkdirSync(dir, { recursive: true }); fs.writeFileSync(path.join(dir, 'manifest.json'), JSON.stringify(plan)); @@ -41,7 +41,7 @@ function report(plan: PaidRunManifest, slices: SliceResult[], collectors: Record fs.writeFileSync(path.join(dir, name), JSON.stringify(body)); } const result = spawnSync(process.execPath, [path.join(ROOT, 'scripts/test-paid-shards.ts'), '--tier', plan.tier, '--report', dir], - { cwd: ROOT, encoding: 'utf8', timeout: 30_000, env: { ...process.env, EVALS_TIER: plan.tier } }); + { cwd: ROOT, encoding: 'utf8', timeout: 30_000, env: { ...process.env, GITHUB_RUN_ID: '', GITHUB_SHA: '', EVALS_TIER: plan.tier, ...env } }); return { status: result.status, out: `${result.stdout}\n${result.stderr}`, dir }; } @@ -219,4 +219,19 @@ describe('behavior and quarantined panels through --report', () => { expect(again.status).toBe(1); expect(again.stdout).toContain('attempt 1 (later attempts 2 reported, never replacing it)'); }); + + test('verdicts become next-run receipts: a whole fresh PASS panel, and negatives for FAIL panels', () => { + const inputKey = 'b'.repeat(64); + const withKey = (results: TrialResult[]) => [1, 2, 3, 4].map(index => slice(index, 4, index === 4 ? [passed(RULE_A)] + : [{ ...trialOutcome(index, results[index - 1]!)!, inputKey }])); + const env = { GITHUB_RUN_ID: '77', GITHUB_SHA: 'e'.repeat(40) }; + const green = report(plan(), withKey(['passed', 'failed', 'passed']), {}, env); + expect(green.status, green.out).toBe(0); + const receipt = JSON.parse(fs.readFileSync(path.join(green.dir, 'report-receipts', `${inputKey}.panel.json`), 'utf8')); + expect(receipt).toMatchObject({ key: inputKey, case: ID, panel: { n: 3, k: 2 }, source: { runId: '77/1' } }); + expect(receipt.trials.map((t: any) => t.outcome)).toEqual(['passed', 'failed', 'passed']); + const red = report(plan(), withKey(['passed', 'failed', 'failed']), {}, env); + expect(red.status).toBe(1); + expect(fs.readdirSync(path.join(red.dir, 'report-receipts'))).toEqual([`${inputKey}.fail.json`]); + }); }); diff --git a/test/paid-run-manifest.test.ts b/test/paid-run-manifest.test.ts index 74090aace..ddd9b96ce 100644 --- a/test/paid-run-manifest.test.ts +++ b/test/paid-run-manifest.test.ts @@ -39,6 +39,7 @@ import { verifySliceResults, expandTrialShards, formatCapacityPreflight, + panelReports, shardSlug, type PaidRunManifest, type ShardOutcome, @@ -607,4 +608,17 @@ describe('trial planner (behavior and quarantined panels)', () => { expect(packed.slices).toHaveLength(3); expect(Object.values(packed.estimates)).toEqual([180_000, 180_000, 180_000]); }); + + test('reuse is whole-panel only: a panel mixing reused and fresh trials is INCOMPLETE', () => { + const manifest = budgetPlan('gate', { 'review-sql-injection': 'behavior' }); + const trials = manifest.entries.filter(entry => entry.trial); + const reused = { inputKey: 'c'.repeat(64), runId: '1001/1', revision: 'd'.repeat(40), completedAt: 1 }; + const results = (reusedTrials: number[]): SliceResult[] => trials.map(entry => ({ version: 1, tier: 'gate', sliceIndex: entry.slice, + sliceCount: manifest.sliceCount, outcomes: [{ files: [entry.file], status: 'passed', exitCode: 0, elapsedMs: 1, executedTests: 1, skippedTests: 0, + trial: { case: 'review-sql-injection', trial: Number(entry.file.slice(-1)), ...entry.trial!, outcome: 'passed', cost_usd: 0, duration_ms: 1 }, + ...(reusedTrials.includes(Number(entry.file.slice(-1))) ? { reused } : {}) }] })); + expect(panelReports(manifest, results([]), 1)[0]).toMatchObject({ status: 'PASS' }); + expect(panelReports(manifest, results([1, 2, 3]), 1)[0]).toMatchObject({ status: 'PASS' }); + expect(panelReports(manifest, results([2]), 1)[0]).toMatchObject({ status: 'INCOMPLETE', failsLane: true, reason: expect.stringContaining('partial panel reuse') }); + }); });