diff --git a/scripts/e2e-shard-reuse.ts b/scripts/e2e-shard-reuse.ts new file mode 100644 index 000000000..692d89250 --- /dev/null +++ b/scripts/e2e-shard-reuse.ts @@ -0,0 +1,186 @@ +/** + * Verified first-attempt reuse for PR-lane E2E shards, on the same receipts as + * the workflow-judge reuse (scripts/eval-input-cache.ts). + * + * A PR-profile shard (a file, or one case of a case-sharded file) is reused + * only when every consumed input is byte-identical to a fresh pass recorded in + * this PR within the receipt age: + * - files: the test file's literal import closure (helpers, fixtures loaded + * as modules, installed packages), every tracked file matched by the + * touchfile patterns of every case the file registers, the global + * touchfiles, the paid runner and this module, the workflow and its setup + * actions, bun.lock and the CI Dockerfile; + * - prompts: the test source that builds each selected case's prompt; + * - parameters: case ids, name pattern, expected count, retries, wall, + * within-shard concurrency, tier/profile, the root package without its + * release label, and every EVALS_/GSTACK_/CLAUDE_/ANTHROPIC_/... variable + * the child receives (secrets contribute presence only); + * - runtime: the immutable CI image manifest, Bun, Node, OS/arch and the + * Claude CLI version. + * Anything unknown fails closed: a computed case registration, a touchfile + * pattern matching no tracked file, retries other than zero (a retried pass + * cannot prove its first attempt), custom preload/endpoints, missing scope, + * or a lane other than the PR gate. Failures are never stored; the weekly + * census, marathon and release lanes always execute fresh. + */ +import * as fs from 'node:fs'; +import * as path from 'node:path'; +import { spawnSync } from 'node:child_process'; +import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache, + type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from './eval-input-cache'; +import { matchGlob } from '../test/helpers/test-selection'; +import { E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from '../test/helpers/touchfiles-data'; + +export interface E2EShardReuseRequest { + root: string; + /** Shard key: `` or `#`. */ + key: string; + file: string; + /** Selected case ids this shard executes (the PR profile's exact expectation). */ + caseIds: string[]; + /** Every E2E id the file registers; all of their touchfiles are consumed inputs. */ + registeredIds: string[]; + registrationKnown: boolean; + casePattern: string; + expectedCases: number; + retries: number; + timeoutMs: number; + withinShardConcurrency: number; + tier: string; + profile: string; + /** The exact environment the child receives. */ + env: NodeJS.ProcessEnv; +} + +export interface E2EShardReuseHit { key: string; source: EvalPassingProof['source'] } + +const HARNESS_FILES = ['scripts/test-paid-shards.ts', 'scripts/e2e-shard-reuse.ts', 'scripts/eval-input-cache.ts', + 'lib/eval-model.ts', 'bun.lock', '.github/docker/Dockerfile.ci', '.github/workflows/evals.yml', + '.github/actions/fix-bun-temp/action.yml', '.github/actions/restore-deps/action.yml', + '.github/actions/seed-claude-config/action.yml', '.github/actions/register-gstack-skills/action.yml']; +const ENV_PREFIXES = ['EVALS_', 'GSTACK_', 'CLAUDE_', 'ANTHROPIC_', 'OPENAI_', 'GEMINI_', 'BUN_', 'NODE_', 'PLAYWRIGHT_']; +/** Run-scoped values: provenance or transport, never behavior. Selection is bound as case ids. */ +const RUN_SCOPED_ENV = new Set(['EVALS_RUN_ID', 'GSTACK_EVAL_DIR', 'EVALS_CACHE_DIR', 'EVALS_CACHE_PR', 'EVALS_CACHE_REPOSITORY', + 'EVALS_CACHE_RUNTIME_ID', 'EVALS_CACHE_PURPOSE', 'EVALS_SELECTION_JSON', 'EVALS_JUDGE_SELECTION_JSON']); +const SECRET_ENV = /KEY|TOKEN|SECRET|PASSWORD|CREDENTIAL/; + +/** The reuse-relevant environment the child sees; secrets contribute presence only. */ +export function e2eReuseEnvironment(env: NodeJS.ProcessEnv): Record { + return Object.fromEntries(Object.keys(env).sort() + .filter(name => (ENV_PREFIXES.some(prefix => name.startsWith(prefix)) || ['PATH', 'HOME'].includes(name)) && !RUN_SCOPED_ENV.has(name)) + .map(name => [name, SECRET_ENV.test(name) ? 'set' : env[name] ?? ''])); +} + +/** Why reuse cannot apply to this lane or environment, else null. */ +export function e2eReuseLaneProblem(env: NodeJS.ProcessEnv, profileMode: string | undefined): string | null { + const pr = Number(env.EVALS_CACHE_PR); + if (profileMode !== 'pr') return 'Only the fast PR profile reuses results'; + if (!env.EVALS_CACHE_DIR || !env.EVALS_CACHE_REPOSITORY || !Number.isSafeInteger(pr) || pr <= 0) return 'No trusted same-PR cache scope'; + if (!/^(?:sha256:)?[a-f0-9]{64}$/.test(env.EVALS_CACHE_RUNTIME_ID ?? '')) return 'No immutable runtime identity'; + if (env.EVALS_TIER !== 'gate' || env.EVALS_FRESH === '1') return 'Fresh validation requested'; + if (['release', 'periodic', 'marathon'].includes(env.EVALS_CACHE_PURPOSE ?? '')) return 'Scheduled and release lanes execute fresh'; + if (env.NODE_OPTIONS || env.BUN_OPTIONS) return 'Preload options change execution outside the consumed source'; + if (env.ANTHROPIC_BASE_URL && env.ANTHROPIC_BASE_URL !== 'https://api.anthropic.com') return 'Custom model endpoint'; + return null; +} + +function trackedFiles(root: string): string[] { + const listed = spawnSync('git', ['ls-files', '-z'], { cwd: root, encoding: 'utf8', timeout: 10_000, maxBuffer: 64 * 1024 * 1024 }); + if (listed.status !== 0) throw new Error('Cannot list tracked files'); + return listed.stdout.split('\0').filter(Boolean); +} + +/** + * Every repository file one shard consumes: the test's import closure plus the + * harness closure, and every tracked file the registered cases' touchfiles and + * the global touchfiles match. Throws when a pattern matches nothing (unknown). + */ +export function e2eShardInputFiles(request: Pick): string[] { + const tracked = trackedFiles(request.root); + const declared = new Set(); + for (const pattern of new Set([...request.registeredIds.flatMap(id => E2E_TOUCHFILES[id] ?? []), ...GLOBAL_TOUCHFILES])) { + const matches = tracked.filter(file => matchGlob(file, pattern)); + if (!matches.length) throw new Error(`Touchfile pattern matches no tracked file: ${pattern}`); + for (const file of matches) declared.add(file); + } + const closure = sourceDependencyClosure(request.root, [request.file, ...HARNESS_FILES]); + return [...new Set([...closure, ...declared])].filter(file => file !== 'package.json').sort(); +} + +/** The consumed input identity of one PR shard, or why it is ineligible. */ +export function e2eShardIdentity(request: E2EShardReuseRequest): { status: 'eligible'; identity: EvalInputIdentity } | { status: 'ineligible'; reason: string } { + try { + if (!/^test\/skill-e2e-.+\.test\.ts$/.test(request.file) || /overlay-harness/.test(request.file)) return { status: 'ineligible', reason: 'Not an audited E2E file' }; + if (!request.registrationKnown || !request.registeredIds.length) return { status: 'ineligible', reason: 'Case registration is not statically complete' }; + if (!request.caseIds.length || request.caseIds.length !== request.expectedCases || request.caseIds.some(id => !request.registeredIds.includes(id))) { + return { status: 'ineligible', reason: 'Selected cases are not exactly known' }; + } + if (request.retries !== 0) return { status: 'ineligible', reason: 'A retried pass cannot prove its first attempt' }; + const files = e2eShardInputFiles(request); + const { version: _releaseLabel, ...rootPackage } = JSON.parse(fs.readFileSync(path.join(request.root, 'package.json'), 'utf8')); + const source = fs.readFileSync(path.join(request.root, request.file), 'utf8'); + const env = request.env; + const result = buildEvalInputIdentity({ + root: request.root, + scope: { repository: env.EVALS_CACHE_REPOSITORY!, pullRequest: Number(env.EVALS_CACHE_PR) }, + coverage: { dependencies: 'complete', prompts: 'complete', environment: 'complete' }, unknownDependencies: [], + files, + prompts: Object.fromEntries(request.caseIds.map(id => [id, source])), + parameters: { rootPackage, key: request.key, caseIds: [...request.caseIds].sort(), casePattern: request.casePattern, + expectedCases: request.expectedCases, retries: request.retries, timeoutMs: request.timeoutMs, + withinShardConcurrency: request.withinShardConcurrency, tier: request.tier, profile: request.profile, + environment: e2eReuseEnvironment(env) }, + runtime: { image: env.EVALS_CACHE_RUNTIME_ID!, bun: Bun.version, node: process.versions.node, + platform: process.platform, arch: process.arch, claudeCli: claudeCliVersion(env) }, + }); + return result; + } catch (error) { + return { status: 'ineligible', reason: error instanceof Error ? error.message : 'Cannot identify consumed inputs' }; + } +} + +function claudeCliVersion(env: NodeJS.ProcessEnv): string { + const version = spawnSync('claude', ['--version'], { encoding: 'utf8', timeout: 5_000, env }); + const line = version.status === 0 ? version.stdout.split('\n')[0]!.trim() : ''; + if (!line) throw new Error('Claude CLI version is unknown'); + return line; +} + +const validResult = (identity: EvalInputIdentity, key: string) => (value: EvalCacheValue) => + !!value && typeof value === 'object' && !Array.isArray(value) + && value.key === key && JSON.stringify(value.cases) === JSON.stringify(identity.caseIds); + +/** + * Prepare reuse for one shard: `lookup` returns a verified receipt for these + * exact inputs; `publish` stores a receipt after a fresh first-attempt pass + * whose inputs did not change during execution. + */ +export function prepareE2EShardReuse(request: E2EShardReuseRequest): { + lookup(): E2EShardReuseHit | null; + publish(): void; +} | null { + if (e2eReuseLaneProblem(request.env, 'pr') !== null) return null; + const before = e2eShardIdentity(request); + if (before.status !== 'eligible') return null; + const common = { cacheDir: request.env.EVALS_CACHE_DIR!, purpose: 'gate' as const }; + return { + lookup() { + const found = lookupEvalInputCache({ ...common, identity: before.identity, validateResult: validResult(before.identity, request.key) }); + return found.status === 'reused' ? { key: found.key, source: found.source } : null; + }, + publish() { + const after = e2eShardIdentity(request); + const env = request.env; + const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID; + const revision = spawnSync('git', ['rev-parse', 'HEAD'], { cwd: request.root, encoding: 'utf8', timeout: 3_000 }); + if (after.status !== 'eligible' || !runId || revision.status !== 0) return; + storeEvalInputCache({ ...common, before: before.identity, after: after.identity, proof: { + execution: 'new', finalized: true, completeAttemptHistory: true, exitCode: 0, timedOut: false, + cancelled: false, skipped: 0, failed: 0, passed: before.identity.caseIds.length, + cases: before.identity.caseIds.map(id => ({ id, outcome: 'passed' as const, attempt: 1 as const })), + source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() }, + result: { key: request.key, cases: before.identity.caseIds }, + } }); + }, + }; +} diff --git a/scripts/eval-input-cache.ts b/scripts/eval-input-cache.ts index e4ecdbb7b..9b489d2bb 100644 --- a/scripts/eval-input-cache.ts +++ b/scripts/eval-input-cache.ts @@ -16,9 +16,50 @@ */ import * as fs from 'node:fs'; import * as path from 'node:path'; +import { isBuiltin } from 'node:module'; import { createHash } from 'node:crypto'; import { atomicWriteSync } from '../lib/fs-atomic'; +/** + * Follow literal module imports from `entries` (repo-relative) without + * executing them, including installed package bytes and the package.json that + * governs each resolved module; root bunfig/tsconfig/jsconfig are included + * when present. The root package.json is left to the caller, which hashes its + * semantic fields without the release version label. + */ +export function sourceDependencyClosure(root: string, entries: string[]): string[] { + const seen = new Set(); + const scan = new Bun.Transpiler({ loader: 'tsx' }); + const visit = (file: string) => { + file = path.resolve(file); + const relative = path.relative(root, file).split(path.sep).join('/'); + if (relative.startsWith('../') || path.isAbsolute(relative)) throw new Error('Dependency outside checkout'); + if (relative === 'package.json') return; + if (seen.has(relative)) return; + seen.add(relative); + const source = fs.readFileSync(file, 'utf8'); + if (!/\.[cm]?[jt]sx?$/.test(file)) return; + // Entrypoint scripts carry hashbangs, which scanImports does not accept. + // Strip only for parsing; buildEvalInputIdentity still hashes the full file. + for (const entry of scan.scanImports(source.replace(/^#![^\n]*(?:\n|$)/, '\n'))) { + if (isBuiltin(entry.path) || entry.path.startsWith('bun:')) continue; + const resolved = Bun.resolveSync(entry.path, path.dirname(file)); + visit(resolved); + // Package export maps/defaults affect resolution independently of code. + let directory = path.dirname(resolved); + while (directory !== root && directory.startsWith(root + path.sep)) { + const manifest = path.join(directory, 'package.json'); + if (fs.existsSync(manifest)) { visit(manifest); break; } + directory = path.dirname(directory); + } + } + }; + for (const file of entries) visit(path.join(root, file)); + for (const file of ['bunfig.toml', 'tsconfig.json', 'jsconfig.json']) + if (fs.existsSync(path.join(root, file))) visit(path.join(root, file)); + return [...seen].sort(); +} + export const EVAL_CACHE_MAX_AGE_MS = 24 * 60 * 60 * 1000; export const EVAL_CACHE_RESULT_MAX_BYTES = 16 * 1024; const SCHEMA = 1; @@ -49,7 +90,7 @@ export type EvalInputIdentityResult = | { status: 'eligible'; identity: EvalInputIdentity } | { status: 'ineligible'; reason: string }; export interface EvalCachePolicy { - purpose: 'gate' | 'periodic' | 'release'; + purpose: 'gate' | 'periodic' | 'release' | 'marathon'; fresh?: boolean; now?: number; maxAgeMs?: number; @@ -146,7 +187,7 @@ function validIdentity(value: unknown): value is EvalInputIdentity { && canonical(value.caseIds) === canonical(sorted(value.caseIds)); } function bypass(policy: EvalCachePolicy): string | null { - if (policy.purpose !== 'gate') return 'Periodic and release validation must execute fresh'; + if (policy.purpose !== 'gate') return 'Periodic, marathon and release validation must execute fresh'; if (policy.fresh) return 'Fresh validation requested'; if (!positive(policy.now ?? Date.now()) || !positive(policy.maxAgeMs ?? EVAL_CACHE_MAX_AGE_MS)) return 'Invalid cache age policy'; return null; diff --git a/test/helpers/workflow-judge-cache.ts b/test/helpers/workflow-judge-cache.ts index db9733e15..0b421567a 100644 --- a/test/helpers/workflow-judge-cache.ts +++ b/test/helpers/workflow-judge-cache.ts @@ -1,13 +1,12 @@ /** Audited cache adapter for runWorkflowJudge only. Native/PTY evals stay fresh. */ import * as fs from 'node:fs'; import * as path from 'node:path'; -import { isBuiltin } from 'node:module'; import { spawnSync } from 'node:child_process'; import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model'; import { JUDGE_MS } from './eval-budgets'; import type { JudgeScore } from './llm-judge'; import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input'; -import { buildEvalInputIdentity, lookupEvalInputCache, storeEvalInputCache, +import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache, type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache'; type Thresholds = { clarity: number; completeness: number; actionability: number }; @@ -25,44 +24,13 @@ export interface WorkflowJudgeReuse { key: string; source: EvalPassingProof['source']; } -/** Follow literal module imports, including installed SDK bytes, without executing them. */ +/** The judge's audited closure: its runner, rubric and documents, installed SDK bytes included. */ export function workflowJudgeDependencies(root: string, documents: string[]): string[] { - const seen = new Set(); - const scan = new Bun.Transpiler({ loader: 'tsx' }); - const visit = (file: string) => { - file = path.resolve(file); - const relative = path.relative(root, file).split(path.sep).join('/'); - if (relative.startsWith('../') || path.isAbsolute(relative)) throw new Error('Dependency outside checkout'); - // Root version labels collector output only; its remaining semantic fields - // are hashed separately. Installed package manifests remain byte-exact. - if (relative === 'package.json') return; - if (seen.has(relative)) return; - seen.add(relative); - const source = fs.readFileSync(file, 'utf8'); - if (!/\.[cm]?[jt]sx?$/.test(file)) return; - // Entrypoint scripts carry hashbangs, which scanImports does not accept. - // Strip only for parsing; buildEvalInputIdentity still hashes the full file. - for (const entry of scan.scanImports(source.replace(/^#![^\n]*(?:\n|$)/, '\n'))) { - if (isBuiltin(entry.path) || entry.path.startsWith('bun:')) continue; - const resolved = Bun.resolveSync(entry.path, path.dirname(file)); - visit(resolved); - // Package export maps/defaults affect resolution independently of code. - let directory = path.dirname(resolved); - while (directory !== root && directory.startsWith(root + path.sep)) { - const manifest = path.join(directory, 'package.json'); - if (fs.existsSync(manifest)) { visit(manifest); break; } - directory = path.dirname(directory); - } - } - }; - for (const file of ['test/skill-llm-eval.test.ts', 'test/helpers/workflow-judge-cache.ts', + return sourceDependencyClosure(root, ['test/skill-llm-eval.test.ts', 'test/helpers/workflow-judge-cache.ts', 'test/helpers/llm-judge.ts', 'lib/eval-model.ts', 'test/helpers/eval-budgets.ts', 'scripts/test-paid-shards.ts', 'scripts/test-strict-output.ts', 'scripts/eval-select.ts', 'scripts/test-pr-profile.ts', '.github/workflows/evals.yml', - 'package.json', 'bun.lock', '.github/docker/Dockerfile.ci', ...documents]) visit(path.join(root, file)); - for (const file of ['bunfig.toml', 'tsconfig.json', 'jsconfig.json']) - if (fs.existsSync(path.join(root, file))) visit(path.join(root, file)); - return [...seen].sort(); + 'package.json', 'bun.lock', '.github/docker/Dockerfile.ci', ...documents]); } export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is JudgeScore & EvalCacheValue {