/** Audited cache adapter for runWorkflowJudge only. Native/PTY evals stay fresh. */ import * as fs from 'node:fs'; import * as path from 'node:path'; import { spawnSync } from 'node:child_process'; import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model'; import { JUDGE_MS } from './eval-budgets'; import type { JudgeScore } from './llm-judge'; import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input'; import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache, type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache'; type Thresholds = { clarity: number; completeness: number; actionability: number }; export interface WorkflowCacheOptions { root: string; testName: string; skillPath: string; startMarker: string; endMarker: string | null; judgeContext: string; judgeGoal: string; model?: string; thresholds: Thresholds; prompt: string; attempt: number; references?: readonly string[]; agentCapability?: 'frontier'; structuredResponse?: boolean; maxTokens?: number; stream?: boolean; env?: NodeJS.ProcessEnv; } export interface WorkflowJudgeReuse { key: string; source: EvalPassingProof['source']; } /** The judge's audited closure: its runner, rubric and documents, installed SDK bytes included. */ export function workflowJudgeDependencies(root: string, documents: string[]): string[] { return sourceDependencyClosure(root, ['test/skill-llm-eval.test.ts', 'test/helpers/workflow-judge-cache.ts', 'test/helpers/llm-judge.ts', 'lib/eval-model.ts', 'test/helpers/eval-budgets.ts', 'scripts/test-paid-shards.ts', 'scripts/test-strict-output.ts', 'scripts/eval-select.ts', 'scripts/test-pr-profile.ts', '.github/workflows/evals.yml', 'package.json', 'bun.lock', '.github/docker/Dockerfile.ci', ...documents]); } export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is JudgeScore & EvalCacheValue { if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).sort().join(',') !== 'actionability,clarity,completeness,reasoning' || typeof value.reasoning !== 'string' || (structuredResponse && (!value.reasoning.trim() || value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false; return (['clarity', 'completeness', 'actionability'] as const).every(key => typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5); } export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): { lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null; /** The attempt guard is rechecked after synchronous input/provenance reads. */ publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined; } { const env = opts.env ?? process.env; const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined }; const pr = Number(env.EVALS_CACHE_PR); // Runtime ID is the immutable CI image manifest, not a mutable image tag. // Nonstandard Node/Bun preload code or custom model endpoints need a separate // audited adapter: they can change the request outside the consumed source. if (!env.EVALS_CACHE_DIR || !env.EVALS_CACHE_REPOSITORY || !Number.isSafeInteger(pr) || pr <= 0 || !/^(?:sha256:)?[a-f0-9]{64}$/.test(env.EVALS_CACHE_RUNTIME_ID ?? '') || env.EVALS_TIER !== 'gate' || env.EVALS_FRESH === '1' || ['release', 'periodic'].includes(env.EVALS_CACHE_PURPOSE ?? '') || opts.attempt !== 1 || env.NODE_OPTIONS || env.BUN_OPTIONS || (env.ANTHROPIC_BASE_URL && env.ANTHROPIC_BASE_URL !== 'https://api.anthropic.com')) return noCache; const currentIdentity = (): EvalInputIdentity | null => { try { const input = readWorkflowJudgeInput(opts); const prompt = buildWorkflowJudgePrompt(opts, input); // Never substitute this adapter's interpretation for the actual API input. if (prompt !== opts.prompt) return null; const { version: _releaseLabel, ...rootPackage } = JSON.parse(fs.readFileSync(path.join(opts.root, 'package.json'), 'utf8')); const identity = buildEvalInputIdentity({ root: opts.root, scope: { repository: env.EVALS_CACHE_REPOSITORY!, pullRequest: pr }, coverage: { dependencies: 'complete', prompts: 'complete', environment: 'complete' }, unknownDependencies: [], files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)), prompts: { [opts.testName]: prompt }, parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS, request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1, ...(opts.stream ? { stream: true } : {}), ...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } }, response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) }, runtime: { image: env.EVALS_CACHE_RUNTIME_ID!, bun: Bun.version, node: process.versions.node, platform: process.platform, arch: process.arch, judge: resolveEvalModel('judge', opts.model, env), anthropic_base_url: env.ANTHROPIC_BASE_URL ?? 'https://api.anthropic.com', anthropic_log: env.ANTHROPIC_LOG ?? null, proxies: Object.fromEntries(['HTTP_PROXY', 'HTTPS_PROXY', 'ALL_PROXY', 'NO_PROXY', 'http_proxy', 'https_proxy', 'all_proxy', 'no_proxy'] .map(name => [name, env[name] ?? null])) }, }); return identity.status === 'eligible' ? identity.identity : null; } catch { return null; } }; const before = currentIdentity(); if (!before) return noCache; const common = { cacheDir: env.EVALS_CACHE_DIR, purpose: 'gate' as const }; return { lookup() { const result = lookupEvalInputCache({ ...common, identity: before, validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) }); return result.status === 'reused' ? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null; }, publish(scores, isActive = () => true) { // Caller reaches here ONLY after its actual assertions passed. A later // failed case in the file does not erase this independently completed case. if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return; const after = currentIdentity(); const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID; if (!after || !runId || !isActive()) return; const revision = spawnSync('git', ['rev-parse', 'HEAD'], { cwd: opts.root, encoding: 'utf8', timeout: 3000 }); if (revision.status !== 0 || !isActive()) return; const stored = storeEvalInputCache({ ...common, before, after, proof: { execution: 'new', finalized: true, completeAttemptHistory: true, exitCode: 0, timedOut: false, cancelled: false, skipped: 0, failed: 0, passed: 1, cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }], source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() }, result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning }, } }); // A slow synchronous write can consume the recording allowance. The // caller withdraws this new receipt if its final deadline check fails. if (stored.status === 'stored') return () => fs.rmSync(path.join(common.cacheDir, `${stored.key}.json`), { force: true }); }, }; }