diff --git a/scripts/typecheck-test-baseline.json b/scripts/typecheck-test-baseline.json index c309682d1..612942cf7 100644 --- a/scripts/typecheck-test-baseline.json +++ b/scripts/typecheck-test-baseline.json @@ -261,7 +261,6 @@ "test/helpers/shared-libs-eval-fixture.ts\tTS7006\tParameter 'candidate' implicitly has an 'any' type.": 2, "test/helpers/shared-libs-path-fixture.ts\tTS2352\tConversion of type '{ root: string; repo: string; state: string; env: { GSTACK_HOME: string; }; }' to type 'SharedLibsFixture' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ root: string; repo: string; state: string; env: { GSTACK_HOME: string; }; }' is missing the following properties from type 'SharedLibsFixture': bin, trace, hookTrace, tip": 1, "test/helpers/shared-libs-plan-actor.ts\tTS18046\t'questions' is of type 'unknown'.": 1, - "test/helpers/workflow-judge-cache.ts\tTS2352\tConversion of type 'string | number | boolean | EvalCacheValue[] | { [key: string]: EvalCacheValue; } | null' to type 'JudgeScore' may be a mistake because neither type sufficiently overlaps with the other. If this was intentional, convert the expression to 'unknown' first. Type '{ [key: string]: EvalCacheValue; }' is missing the following properties from type 'JudgeScore': clarity, completeness, actionability, reasoning": 1, "test/impeccable-fixtures.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type 'string | undefined' is not assignable to parameter of type 'string'. Type 'undefined' is not assignable to type 'string'.": 1, "test/llm-judge-abort.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type '{ passed: boolean; }' is not assignable to parameter of type 'undefined'.": 3, "test/llm-judge-frontier.test.ts\tTS2769\tNo overload matches this call. The last overload gave the following error. Argument of type '{ score: number; reason: string; }' is not assignable to parameter of type 'undefined'.": 1, diff --git a/test/cookie-workflow-judge-input.test.ts b/test/cookie-workflow-judge-input.test.ts index cc174995e..70e2b489c 100644 --- a/test/cookie-workflow-judge-input.test.ts +++ b/test/cookie-workflow-judge-input.test.ts @@ -11,7 +11,7 @@ import { selectTests } from './helpers/test-selection'; import { E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data'; import { selectPrProfile } from '../scripts/test-pr-profile'; import { JUDGE_MS } from './helpers/eval-budgets'; -import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge'; +import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge'; import { COOKIE_MANUAL_REVIEW_FILE, getCookieWorkflowManualReview, isManualReviewEntry } from './helpers/cookie-workflow-manual-review'; const ROOT = resolve(import.meta.dir, '..'); @@ -70,7 +70,7 @@ function actualCookieCallback(root: string, overrides: { const records: EvalTestEntry[] = []; const attempts = new Map(); let callback: () => Promise = async () => { throw new Error('Judge callback was not registered'); }; - new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', registration)( + new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', 'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS', registration)( (_suite: string, names: string[], run: () => void) => { expect(names).toEqual([NAME]); run(); }, (name: string, run: () => Promise, budget: number) => { expect(name).toBe(NAME); expect(budget).toBe(JUDGE_MS + 10_000); callback = run; }, root, buildCookieWorkflowJudgeInput, (_kind: string, explicit?: string) => explicit ?? 'fixture-model', @@ -85,7 +85,7 @@ function actualCookieCallback(root: string, overrides: { attempts, overrides.clock ? { now: overrides.clock } : performance, overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout, JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS, - WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore, + WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, ); return { run: () => callback(), requests, records, attempts }; } @@ -96,7 +96,7 @@ describe('cookie workflow judge input', () => { approveFixture(root); const h = actualCookieCallback(root, { judge: async () => { throw refusal(); } }); await h.run(); - expect(h.requests).toHaveLength(1); + expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES); expect(h.records).toHaveLength(1); expect(h.records[0]).toMatchObject({ passed: false, execution: 'executed', exit_reason: 'provider_refusal' }); expect(isManualReviewEntry(h.records[0])).toBe(true); @@ -161,7 +161,7 @@ describe('cookie workflow judge input', () => { const root = fixture(); approveFixture(root); let calls = 0; const h = actualCookieCallback(root, { judge: async () => { - if (++calls === 1) return { ...passingScore, clarity: 1 }; + if (++calls <= JUDGE_PANEL_SAMPLES) return { ...passingScore, clarity: 1 }; throw refusal(); } }); await expect(h.run()).rejects.toThrow(); @@ -275,7 +275,7 @@ describe('cookie workflow judge input', () => { let scores = passingScore; const h = actualCookieCallback(root, { judge: async () => scores }); await h.run(); - expect(h.requests).toHaveLength(1); + expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES); expect(h.requests[0].prompt).toBe(input.prompt); expect(h.requests[0].model).toBe(COOKIE_WORKFLOW_JUDGE.model); expect(h.requests[0].signal).toBeInstanceOf(AbortSignal); @@ -283,7 +283,7 @@ describe('cookie workflow judge input', () => { expect(existsSync(join(root, 'cache'))).toBe(false); const fresh = actualCookieCallback(root); await fresh.run(); - expect(fresh.requests).toHaveLength(1); + expect(fresh.requests).toHaveLength(JUDGE_PANEL_SAMPLES); for (const dimension of ['clarity', 'completeness', 'actionability'] as const) { scores = { ...COOKIE_WORKFLOW_JUDGE.thresholds, [dimension]: COOKIE_WORKFLOW_JUDGE.thresholds[dimension] - 1, reasoning: 'Synthetic failing fixture score' }; await expect(h.run()).rejects.toThrow(); diff --git a/test/helpers/llm-judge.ts b/test/helpers/llm-judge.ts index 4a0f72a1c..da7811b47 100644 --- a/test/helpers/llm-judge.ts +++ b/test/helpers/llm-judge.ts @@ -23,6 +23,8 @@ export interface JudgeScore { reasoning: string; } +export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const; + export interface JudgeRefusalEvidence { stop_reason: 'refusal'; response_id: string | null; @@ -196,6 +198,63 @@ export async function callJudge( } } +/** + * Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this + * helper (imported by many paid tests) does not pull the quarantine registry + * into their touchfile closure. test/judge-panel.test.ts pins the two equal. + */ +export const JUDGE_PANEL_SAMPLES = 3; + +/** + * Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of + * independent samples of the SAME prompt concurrently, inside its unchanged + * JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean + * against the unchanged minimum; boolean fields gate on a strict majority. + * A sample that errors (refusal, truncation, non-JSON, malformed field) fails + * the whole panel and is never resampled. callJudge's 429 backoff happens + * before any model output exists, so it is transport, not a verdict retry. + */ +export async function judgePanel(sample: () => Promise): Promise { + const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample())); + const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []); + if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult).value); + const first = failures[0]!; + // A refusal is an unscored panel only when EVERY sample refused; a partial + // refusal beside scored samples is an ordinary failed panel. + if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) { + throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`); + } + throw first.reason; +} + +/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */ +export function judgePanelMean(samples: ReadonlyArray>, keys: readonly K[]): Record { + if (samples.length === 0) throw new Error('Judge panel has no samples'); + return Object.fromEntries(keys.map(key => { + const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined); + const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value)); + if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`); + return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length]; + })) as Record; +} + +/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */ +export function judgePanelMajority(samples: ReadonlyArray>, key: K): boolean { + if (samples.length === 0) throw new Error('Judge panel has no samples'); + const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined); + const bad = values.findIndex(value => typeof value !== 'boolean'); + if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`); + return values.filter(value => value === true).length * 2 > values.length; +} + +/** Sample reasoning lines, numbered, for the collector record. */ +export function judgePanelReasoning(samples: ReadonlyArray): string { + return samples.map((sample, index) => { + const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined; + return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`; + }).join('\n'); +} + /** * Score documentation quality on clarity/completeness/actionability (1-5). */ diff --git a/test/helpers/periodic-exclude-data.ts b/test/helpers/periodic-exclude-data.ts index 5959f44d7..5628e13d7 100644 --- a/test/helpers/periodic-exclude-data.ts +++ b/test/helpers/periodic-exclude-data.ts @@ -72,7 +72,12 @@ export const CASE_CI_EXCLUDE: Record= `exit.rate` over >= * `exit.minTrials`; at most `capFraction` of each tier's * blocking cases; an entry expires after `expiryWeeklyRuns`. - * drift - one-sided Fisher exact alarm between input-identity series. + * judge - a judge case draws `samples` independent samples of one + * prompt concurrently; numeric dimensions gate on the panel + * mean against the unchanged threshold, booleans on a strict + * majority; an erroring sample fails the panel, never resampled. + * drift - one-sided Fisher exact alarm between input-identity series + * (Holm-controlled across the cases tested in one report). * infraRedispatch - a census whose every red verdict is machine-classified * INFRA or INCOMPLETE may be re-dispatched this many times as * a new run; both runs are reported. @@ -86,6 +91,7 @@ export const EVAL_POLICY = { capFraction: 0.10, expiryWeeklyRuns: 8, }, + judge: { samples: 3 }, drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 }, infraRedispatch: 1, } as const; diff --git a/test/helpers/workflow-judge-cache.ts b/test/helpers/workflow-judge-cache.ts index 0b421567a..ee9607546 100644 --- a/test/helpers/workflow-judge-cache.ts +++ b/test/helpers/workflow-judge-cache.ts @@ -4,7 +4,7 @@ import * as path from 'node:path'; import { spawnSync } from 'node:child_process'; import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model'; import { JUDGE_MS } from './eval-budgets'; -import type { JudgeScore } from './llm-judge'; +import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge'; import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input'; import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache, type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache'; @@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres || typeof value.reasoning !== 'string' || (structuredResponse && (!value.reasoning.trim() || value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false; - return (['clarity', 'completeness', 'actionability'] as const).every(key => + return JUDGE_SCORE_DIMENSIONS.every(key => typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5); } +const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 }; + +/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */ +export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array } { + if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples' + || !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES + || !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false; + const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS); + return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]); +} + export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): { - lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null; + lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null; /** The attempt guard is rechecked after synchronous input/provenance reads. */ - publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined; + publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined; } { const env = opts.env ?? process.env; - const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined }; + const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined }; const pr = Number(env.EVALS_CACHE_PR); // Runtime ID is the immutable CI image manifest, not a mutable image tag. // Nonstandard Node/Bun preload code or custom model endpoints need a separate @@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): { files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)), prompts: { [opts.testName]: prompt }, parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS, - request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1, + request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0, + panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' }, ...(opts.stream ? { stream: true } : {}), ...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } }, response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) }, @@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): { return { lookup() { const result = lookupEvalInputCache({ ...common, identity: before, - validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) }); + validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) }); return result.status === 'reused' - ? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null; + ? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null; }, - publish(scores, isActive = () => true) { + publish(samples, isActive = () => true) { // Caller reaches here ONLY after its actual assertions passed. A later // failed case in the file does not erase this independently completed case. - if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return; + const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) }; + if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return; const after = currentIdentity(); const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID; if (!after || !runId || !isActive()) return; @@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): { cancelled: false, skipped: 0, failed: 0, passed: 1, cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }], source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() }, - result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning }, + result: panel, } }); // A slow synchronous write can consume the recording allowance. The // caller withdraws this new receipt if its final deadline check fails. diff --git a/test/skill-llm-eval.test.ts b/test/skill-llm-eval.test.ts index 41b5ab13c..af9d1774c 100644 --- a/test/skill-llm-eval.test.ts +++ b/test/skill-llm-eval.test.ts @@ -14,7 +14,7 @@ import { afterAll, expect } from 'bun:test'; import { JUDGE_MS } from './helpers/eval-budgets'; import * as fs from 'fs'; import * as path from 'path'; -import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge'; +import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelMajority, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS } from './helpers/llm-judge'; import { ENG_REVIEW_EXCERPT } from './helpers/workflow-excerpt'; import type { JudgeScore } from './helpers/llm-judge'; import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA, type WorkflowJudgeInput } from './helpers/workflow-judge-input'; @@ -100,8 +100,9 @@ describeIfSelected('LLM-as-judge quality evals', [ // rewrites the pin). const section = sliceBrowseSection('## Snapshot Flags'); - const scores = await judge('browse skill reference (flags + commands)', section); - console.log('Browse SKILL.md scores:', JSON.stringify(scores, null, 2)); + const samples = await judgePanel(() => judge('browse skill reference (flags + commands)', section)); + const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS); + console.log('Browse SKILL.md panel:', JSON.stringify({ mean: scores, samples }, null, 2)); const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json'); const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8')); @@ -120,9 +121,9 @@ describeIfSelected('LLM-as-judge quality evals', [ tier: 'llm-judge', passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4 && regressions.length === 0, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: regressions.length ? `${scores.reasoning} | ${regressions.join('; ')}` : scores.reasoning, + judge_reasoning: regressions.length ? `${judgePanelReasoning(samples)} | ${regressions.join('; ')}` : judgePanelReasoning(samples), }); expect(scores.clarity).toBeGreaterThanOrEqual(3); @@ -144,8 +145,9 @@ describeIfSelected('LLM-as-judge quality evals', [ if (setupStart < 0 || setupEnd < 0) throw new Error('browse/SKILL.md: setup block not found — regenerate with: bun run gen:skill-docs'); const section = content.slice(setupStart, setupEnd); - const scores = await judge('setup/binary discovery instructions', section); - console.log('Setup block scores:', JSON.stringify(scores, null, 2)); + const samples = await judgePanel(() => judge('setup/binary discovery instructions', section)); + const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS); + console.log('Setup block panel:', JSON.stringify({ mean: scores, samples }, null, 2)); evalCollector?.addTest({ name: 'setup block', @@ -153,9 +155,9 @@ describeIfSelected('LLM-as-judge quality evals', [ tier: 'llm-judge', passed: scores.actionability >= 3 && scores.clarity >= 3, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); // Setup block is intentionally minimal (binary discovery only). @@ -203,7 +205,7 @@ describeIfSelected('QA skill quality evals', ['qa/SKILL.md workflow', 'qa/SKILL. startMarker: '# /qa: Test', endMarker: null, references: ['qa/templates/functional-report-template.md'] }).text; - const scores = await callJudge(`You are evaluating the quality of a QA testing workflow document for an AI coding agent. + const samples = await judgePanel(() => callJudge(`You are evaluating the quality of a QA testing workflow document for an AI coding agent. The agent reads this source-file bundle to select browser, native functional or mixed surfaces, explore with bounded probes, reproduce and diagnose defects, add a regression @@ -222,8 +224,9 @@ Respond with ONLY valid JSON: Here is the QA workflow to evaluate: -${section}`); - console.log('QA workflow scores:', JSON.stringify(scores, null, 2)); +${section}`)); + const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS); + console.log('QA workflow panel:', JSON.stringify({ mean: scores, samples }, null, 2)); evalCollector?.addTest({ name: 'qa/SKILL.md workflow', @@ -231,9 +234,9 @@ ${section}`); tier: 'llm-judge', passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); expect(scores.clarity).toBeGreaterThanOrEqual(3); @@ -247,7 +250,7 @@ ${section}`); const t0 = Date.now(); const section = sliceQaPatterns('## Health Score Rubric'); - const scores = await callJudge(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score. + const samples = await judgePanel(() => callJudge(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score. The agent uses this rubric after QA testing a website. It needs to: 1. Understand each scoring category and what counts as a deduction @@ -264,8 +267,9 @@ Respond with ONLY valid JSON: Here is the rubric to evaluate: -${section}`); - console.log('QA health rubric scores:', JSON.stringify(scores, null, 2)); +${section}`)); + const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS); + console.log('QA health rubric panel:', JSON.stringify({ mean: scores, samples }, null, 2)); evalCollector?.addTest({ name: 'qa/SKILL.md health rubric', @@ -273,9 +277,9 @@ ${section}`); tier: 'llm-judge', passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); expect(scores.clarity).toBeGreaterThanOrEqual(3); @@ -294,7 +298,7 @@ ${section}`); const diffAwareSection = sliceQaPatterns('### Diff-aware', '### Full'); const rulesSection = sliceQaPatterns('## Important Rules'); - const result = await callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario. + const samples = await judgePanel(() => callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario. SCENARIO: A user runs /qa (a browser-based QA testing skill). The branch diff shows ONLY prompt template files and config file changes — no routes, views, controllers, components, or CSS were changed. The changes are "purely backend" with no obvious UI surface. @@ -318,9 +322,10 @@ Respond with ONLY valid JSON: Rules: - would_browse should be true if the document instructs the agent to always use the browser regardless of diff content - would_browse should be false if the document allows the agent to skip browser testing for non-UI changes -- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`); +- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`)); + const result = { would_browse: judgePanelMajority(samples, 'would_browse'), ...judgePanelMean(samples, ['confidence'] as const) }; - console.log('QA anti-refusal result:', JSON.stringify(result, null, 2)); + console.log('QA anti-refusal panel:', JSON.stringify({ result, samples }, null, 2)); evalCollector?.addTest({ name: 'qa/SKILL.md anti-refusal', @@ -328,9 +333,9 @@ Rules: tier: 'llm-judge', passed: result.would_browse === true && result.confidence >= 4, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { would_browse: result.would_browse ? 1 : 0, confidence: result.confidence }, - judge_reasoning: result.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); expect(result.would_browse).toBe(true); @@ -362,7 +367,7 @@ describeIfSelected('Cross-skill consistency evals', ['cross-skill greptile consi extractGrepLines(retroContent, 'retro/SKILL.md'), ].join('\n\n'); - const result = await callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently. + const samples = await judgePanel(() => callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently. INTENDED ARCHITECTURE: - greptile-history has TWO paths: per-project (~/.gstack/projects/{slug}/greptile-history.md) and global (~/.gstack/greptile-history.md) @@ -383,9 +388,10 @@ Evaluate consistency. Respond with ONLY valid JSON: "reasoning": "brief explanation" } -score (1-5): 5 = perfectly consistent, 1 = contradictory`); +score (1-5): 5 = perfectly consistent, 1 = contradictory`)); + const result = { consistent: judgePanelMajority(samples, 'consistent'), ...judgePanelMean(samples, ['score'] as const) }; - console.log('Cross-skill consistency:', JSON.stringify(result, null, 2)); + console.log('Cross-skill consistency panel:', JSON.stringify({ result, samples }, null, 2)); evalCollector?.addTest({ name: 'cross-skill greptile consistency', @@ -393,9 +399,9 @@ score (1-5): 5 = perfectly consistent, 1 = contradictory`); tier: 'llm-judge', passed: result.consistent && result.score >= 4, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { consistency_score: result.score }, - judge_reasoning: result.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); expect(result.consistent).toBe(true); @@ -439,7 +445,8 @@ async function runWorkflowJudge(opts: { const workDeadline = started + JUDGE_MS; let stage: 'input' | 'judge' | 'validation' | 'recording' = 'input'; let finalized = false; - let scores: JudgeScore | undefined; + let samples: JudgeScore[] | undefined; + let scores: Record | undefined; let manualReview: ManualJudgeReview | undefined; let customInputMetadata: { prompt: string; model: string } | undefined; let reused: ReturnType['lookup']> = null; @@ -458,19 +465,19 @@ async function runWorkflowJudge(opts: { evalCollector?.addTest({ name: opts.testName, suite: opts.suite, tier: 'llm-judge', passed, attempt, duration_ms: Math.max(0, performance.now() - started), - cost_usd: reused || !scores ? 0 : 0.02, + cost_usd: reused || !samples ? 0 : 0.02 * samples.length, execution: reused ? 'reused' : 'executed', ...customInputMetadata, ...(manualReview ? { manual_review: manualReview } : {}), ...(reused ? { reused_from: { input_key: reused.reuse.key, run_id: reused.reuse.source.runId, revision: reused.reuse.source.revision, completed_at: new Date(reused.reuse.source.completedAt).toISOString() } } : {}), - ...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning } : {}), + ...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability } } : {}), + ...(samples ? { judge_reasoning: judgePanelReasoning(samples) } : {}), ...(passed ? {} : { exit_reason: error instanceof JudgeRefusalError ? 'provider_refusal' : error instanceof Error && error.name === 'WorkflowJudgeDeadline' ? 'timeout' : error instanceof Error && error.name === 'WorkflowJudgeSuperseded' ? 'cancelled' : stage === 'validation' ? 'validation_failed' : 'harness_error', - error: `${error instanceof Error ? error.message : String(error)}${scores ? '' : error instanceof JudgeRefusalError + error: `${error instanceof Error ? error.message : String(error)}${samples ? '' : error instanceof JudgeRefusalError ? '\nNo automated score; provider refusal usage retained when manually accepted; cost unavailable.' : '\nNo completed model response; cost and usage unavailable.'}` }), }); @@ -508,11 +515,11 @@ async function runWorkflowJudge(opts: { checkActive(); stage = 'judge'; const maxTokens = opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS; - let result: JudgeScore; + let result: JudgeScore[]; try { - result = reused?.scores ?? await callJudge(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens, + result = reused?.samples ?? await judgePanel(() => callJudge(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens, ...(opts.stream ? { stream: true } : {}), - ...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) }); + ...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) })); } catch (error) { checkActive(); if (error instanceof JudgeRefusalError && customInputMetadata) { @@ -529,20 +536,21 @@ async function runWorkflowJudge(opts: { throw error; } checkActive(); - scores = result; + samples = result; console.log(`[workflow-judge] ${opts.testName}: ${reused ? `reused ${reused.reuse.source.runId} @ ${reused.reuse.source.revision} (${new Date(reused.reuse.source.completedAt).toISOString()})` : 'executed'}`); - console.log(`${opts.testName} scores:`, JSON.stringify(scores, null, 2)); stage = 'validation'; - if (opts.structuredResponse && !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true)) { + if (opts.structuredResponse && !samples.every(sample => validWorkflowJudgeScore(sample as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true))) { throw new Error('Structured workflow judge violated the response schema'); } + scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS); + console.log(`${opts.testName} panel:`, JSON.stringify({ mean: scores, samples }, null, 2)); expect(scores.clarity).toBeGreaterThanOrEqual(thresholds.clarity); expect(scores.completeness).toBeGreaterThanOrEqual(thresholds.completeness); expect(scores.actionability).toBeGreaterThanOrEqual(thresholds.actionability); checkActive(); stage = 'recording'; arm(); - const discardReceipt = reused ? undefined : cache.publish(scores, active); + const discardReceipt = reused ? undefined : cache.publish(samples, active); try { checkActive(); finish(true); } catch (error) { discardReceipt?.(); throw error; } }; @@ -792,7 +800,7 @@ describeIfSelected('Voice directive eval', ['voice directive tone'], () => { const voiceEnd = content.indexOf('\n## ', voiceStart + 1); const voiceSection = content.slice(voiceStart, voiceEnd > 0 ? voiceEnd : voiceStart + 3000); - const result = await callJudge<{ + const samples = await judgePanel(() => callJudge<{ directness: number; concreteness: number; avoids_corporate: number; @@ -812,9 +820,10 @@ Return JSON only: {"directness": N, "concreteness": N, "avoids_corporate": N, "avoids_ai_vocabulary": N, "connects_user_outcomes": N, "reasoning": "..."} THE VOICE DIRECTIVE: -${voiceSection}`); +${voiceSection}`)); + const result = judgePanelMean(samples, ['directness', 'concreteness', 'avoids_corporate', 'avoids_ai_vocabulary', 'connects_user_outcomes'] as const); - console.log('Voice directive scores:', JSON.stringify(result, null, 2)); + console.log('Voice directive panel:', JSON.stringify({ mean: result, samples }, null, 2)); evalCollector?.addTest({ name: 'voice directive tone', @@ -823,7 +832,7 @@ ${voiceSection}`); passed: result.directness >= 4 && result.concreteness >= 4 && result.avoids_corporate >= 4 && result.avoids_ai_vocabulary >= 4 && result.connects_user_outcomes >= 4, duration_ms: Date.now() - t0, - cost_usd: 0.02, + cost_usd: 0.02 * samples.length, judge_scores: { directness: result.directness, concreteness: result.concreteness, @@ -831,7 +840,7 @@ ${voiceSection}`); avoids_ai_vocabulary: result.avoids_ai_vocabulary, connects_user_outcomes: result.connects_user_outcomes, }, - judge_reasoning: result.reasoning, + judge_reasoning: judgePanelReasoning(samples), }); expect(result.directness).toBeGreaterThanOrEqual(4); diff --git a/test/workflow-judge-cache.test.ts b/test/workflow-judge-cache.test.ts index 79feebcdd..1d55fad33 100644 --- a/test/workflow-judge-cache.test.ts +++ b/test/workflow-judge-cache.test.ts @@ -1,18 +1,22 @@ -import { afterEach, expect, spyOn, test } from 'bun:test'; +import { afterEach, describe, expect, spyOn, test } from 'bun:test'; import { Messages } from '@anthropic-ai/sdk/resources/messages'; -import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge'; +import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMajority, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge'; +import { EVAL_POLICY } from './helpers/periodic-exclude-data'; import { getCookieWorkflowManualReview } from './helpers/cookie-workflow-manual-review'; import { resolveEvalModel } from '../lib/eval-model'; import * as fs from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; import { execFileSync } from 'node:child_process'; -import { prepareWorkflowJudgeCache, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache'; +import { prepareWorkflowJudgeCache, validWorkflowJudgePanel, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache'; import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA } from './helpers/workflow-judge-input'; const roots: string[] = []; afterEach(() => { for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); }); const scores = { clarity: 4, completeness: 5, actionability: 4, reasoning: 'Concrete steps' }; +const SAMPLES = JUDGE_PANEL_SAMPLES; +const panelOf = (sample: typeof scores) => Array.from({ length: SAMPLES }, () => sample); +const panel = panelOf(scores); function fixture() { const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-judge-cache-')); roots.push(root); const files = { @@ -53,9 +57,9 @@ function fixture() { } test('the audited adapter reuses only the exact completed score and original provenance', () => { - const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(scores); + const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(panel); expect(f.entries()).toHaveLength(1); - const reused = f.cache().lookup(); expect(reused?.scores).toEqual(scores); + const reused = f.cache().lookup(); expect(reused?.samples).toEqual(panel); expect(reused?.reuse.source.runId).toBe('free-cache-test'); expect(reused?.reuse.source.revision).toMatch(/^[a-f0-9]{40}$/); expect(reused?.reuse.source.completedAt).toBeLessThanOrEqual(Date.now()); @@ -73,11 +77,11 @@ test('the dependency closure includes actual installed SDK bytes and local trans }); test('release-label changes preserve reuse; other package semantics invalidate it', () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); const file = path.join(f.root, 'package.json'); const original = JSON.parse(fs.readFileSync(file, 'utf8')); fs.writeFileSync(file, JSON.stringify({ ...original, version: '2.0.0' }, null, 2)); - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); for (const change of [{ scripts: { 'test:gate': 'changed command' } }, { dependencies: { 'some-sdk': '2.0.0' } }]) { fs.writeFileSync(file, JSON.stringify({ ...original, ...change, version: '2.0.0' })); expect(f.cache().lookup()).toBeNull(); @@ -89,7 +93,7 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in 'scripts/test-paid-shards.ts', 'scripts/test-strict-output.ts', 'scripts/eval-select.ts', 'scripts/test-pr-profile.ts', '.github/workflows/evals.yml']) { test(`changes in ${file} require new evaluation`, () => { - const f = fixture(); f.cache().publish(scores); const target = path.join(f.root, file); + const f = fixture(); f.cache().publish(panel); const target = path.join(f.root, file); fs.appendFileSync(target, file.endsWith('.json') ? ' ' : '\n// changed'); f.refreshPrompt(); expect(f.cache().lookup()).toBeNull(); }); @@ -98,8 +102,8 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in test('changed sources during an attempt and mismatched actual prompt cannot publish', () => { const f = fixture(); const before = f.cache(); fs.appendFileSync(path.join(f.root, 'example/sections/review.md'), 'new finding'); - before.publish(scores); expect(f.entries()).toHaveLength(0); - f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(scores); + before.publish(panel); expect(f.entries()).toHaveLength(0); + f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(panel); expect(f.entries()).toHaveLength(0); }); @@ -108,52 +112,52 @@ for (const [key, value] of Object.entries({ EVALS_FRESH: '1', EVALS_TIER: 'perio EVALS_CACHE_REPOSITORY: '', NODE_OPTIONS: '--require=unknown', BUN_OPTIONS: '--preload=unknown', ANTHROPIC_BASE_URL: 'https://custom-provider.example.test' })) { test(`${key}=${value} is fresh or ineligible`, () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); f.opts.env = { ...f.env, [key]: value }; const cache = f.cache(); - expect(cache.lookup()).toBeNull(); cache.publish(scores); expect(f.entries()).toHaveLength(1); + expect(cache.lookup()).toBeNull(); cache.publish(panel); expect(f.entries()).toHaveLength(1); }); } test('runtime/model/threshold changes miss, and retries never reuse or publish', () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); for (const overrides of [{ GSTACK_EVAL_MODEL_JUDGE: 'different-model' }, { EVALS_CACHE_RUNTIME_ID: 'c'.repeat(64) }]) { f.opts.env = { ...f.env, ...overrides }; expect(f.cache().lookup()).toBeNull(); } f.opts.env = f.env; f.opts.thresholds.clarity = 5; expect(f.cache().lookup()).toBeNull(); f.opts.thresholds.clarity = 4; f.opts.attempt = 2; const retry = f.cache(); - expect(retry.lookup()).toBeNull(); retry.publish(scores); expect(f.entries()).toHaveLength(1); + expect(retry.lookup()).toBeNull(); retry.publish(panel); expect(f.entries()).toHaveLength(1); }); test('frontier reader calibration cannot reuse a score from the unspecified-reader rubric', () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); const original = f.opts.prompt; f.opts.agentCapability = 'frontier'; f.refreshPrompt(); expect(f.opts.prompt).not.toBe(original); expect(f.cache().lookup()).toBeNull(); - f.cache().publish(scores); + f.cache().publish(panel); expect(f.entries()).toHaveLength(2); - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); delete f.opts.agentCapability; f.refreshPrompt(); expect(f.opts.prompt).toBe(original); - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); }); test('a pinned workflow judge model overrides the global model and changes the cache identity', () => { const f = fixture(); f.opts.model = 'claude-sonnet-4-6'; - f.cache().publish(scores); + f.cache().publish(panel); expect(f.entries()).toHaveLength(1); f.opts.env = { ...f.env, GSTACK_EVAL_MODEL_JUDGE: 'different-global-model' }; - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); f.opts.model = 'claude-opus-4-7'; expect(f.cache().lookup()).toBeNull(); }); test('failed assertions, missing provenance, and missing imported dependencies cannot supply a receipt', () => { - const f = fixture(); f.cache().publish({ ...scores, clarity: 3 }); expect(f.entries()).toHaveLength(0); - f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(scores); expect(f.entries()).toHaveLength(0); + const f = fixture(); f.cache().publish(panelOf({ ...scores, clarity: 3 })); expect(f.entries()).toHaveLength(0); + f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(panel); expect(f.entries()).toHaveLength(0); f.opts.env = f.env; fs.unlinkSync(path.join(f.root, 'test/helpers/nested.ts')); - f.cache().publish(scores); expect(f.entries()).toHaveLength(0); + f.cache().publish(panel); expect(f.entries()).toHaveLength(0); }); test('cached payload schema remains small and cannot carry operational fields', () => { @@ -167,8 +171,9 @@ test('workflow registration preserves model work and reserves only terminal-reco const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8'); const body = source.split('async function runWorkflowJudge')[1]!.split('// Block 1:')[0]!; const stages = ['workflowJudgeAttempts.set', 'readWorkflowJudgeInput(', 'cache.lookup()', - 'callJudge(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,', - 'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(scores, active)'] + 'judgePanel(() => callJudge(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,', + 'scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);', + 'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(samples, active)'] .map(stage => body.indexOf(stage)); expect(stages.every(position => position >= 0)).toBe(true); expect(stages).toEqual([...stages].sort((a, b) => a - b)); @@ -203,6 +208,7 @@ function actualCallback(f: ReturnType, overrides: { 'evalCollector', 'expect', 'console', 'performance', 'JUDGE_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'resolveEvalModel', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', + 'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS', `${javascript}\nreturn runWorkflowJudge;`)( f.root, overrides.read ?? readWorkflowJudgeInput, buildWorkflowJudgePrompt, (options: WorkflowCacheOptions) => (overrides.prepare ?? prepareWorkflowJudgeCache)({ ...options, env: f.env }), @@ -213,7 +219,8 @@ function actualCallback(f: ReturnType, overrides: { overrides.clock ? { now: overrides.clock } : performance, overrides.budget ?? 120_000, overrides.allowance ?? 5_000, overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout, JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel, - WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore); + WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore, + judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS); return { run, records, signals, prompts, attempts, options: { ...f.opts, suite: 'Cache regression' } }; } @@ -223,7 +230,7 @@ test('the actual workflow callback preserves the pinned model and frontier rubri const actual = actualCallback(f, { judge: async (_prompt, model) => { models.push(model); return scores; } }); await actual.run({ ...actual.options, model: 'claude-sonnet-4-6', agentCapability: 'frontier', readInput: () => readWorkflowJudgeInput(f.opts) }); - expect(models).toEqual(['claude-sonnet-4-6']); + expect(models).toEqual(Array(SAMPLES).fill('claude-sonnet-4-6')); expect(actual.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger'); expect(actual.records[0]).toMatchObject({ passed: true, model: 'claude-sonnet-4-6', prompt: actual.prompts[0] }); }); @@ -239,7 +246,7 @@ test.each(['ship', 'review'])('the registered %s callback sends the frontier rub endMarker: f.opts.endMarker, references: [] }; const passing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 3 }) }); await passing.run(options); - expect(passing.prompts).toHaveLength(1); + expect(passing.prompts).toHaveLength(SAMPLES); expect(passing.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger'); expect(passing.records[0]).toMatchObject({ passed: true, execution: 'executed', judge_scores: { clarity: 3 } }); const failing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 2 }) }); @@ -254,8 +261,8 @@ test('the actual workflow callback executes once, reuses with provenance, and pr const f = fixture(); const first = actualCallback(f); const options = { ...f.opts, suite: 'Cache regression' }; await first.run(options); - expect(first.prompts).toEqual([f.opts.prompt]); expect(f.entries()).toHaveLength(1); - expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 }); + expect(first.prompts).toEqual(Array(SAMPLES).fill(f.opts.prompt)); expect(f.entries()).toHaveLength(1); + expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 * SAMPLES }); expect(first.records[0]).not.toHaveProperty('prompt'); expect(first.records[0]).not.toHaveProperty('model'); const reused = actualCallback(f, { judge: async () => ({ ...scores, clarity: 1 }) }); @@ -312,13 +319,13 @@ test('a superseding attempt cancels its predecessor before either can record a s }); test('a failed input read consumes attempt one and prevents a retry from borrowing or publishing a receipt', async () => { - const f = fixture(); f.cache().publish(scores); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8'); + const f = fixture(); f.cache().publish(panel); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8'); let reads = 0; const h = actualCallback(f, { read: options => { if (++reads === 1) throw new Error('Missing workflow fixture'); return readWorkflowJudgeInput(options); } }); await expect(h.run(h.options)).rejects.toThrow('Missing workflow fixture'); expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'harness_error' }); await h.run(h.options); - expect(h.prompts).toHaveLength(1); + expect(h.prompts).toHaveLength(SAMPLES); expect(h.records.map(record => record.execution)).toEqual(['executed', 'executed']); expect(h.attempts.get(f.opts.testName).attempt).toBe(2); expect(fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8')).toBe(receipt); @@ -333,14 +340,14 @@ test('monotonic expiry after a synchronous preparation or late model response re await expect(h.run(h.options)).rejects.toThrow('deadline'); expect(h.records).toHaveLength(1); expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'timeout', duration_ms: 21 }); - expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : 1); + expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : SAMPLES); expect(f.entries()).toHaveLength(0); } }); test('publication rechecks after input scanning and withdraws a receipt if recording expires', async () => { const f = fixture(); let checks = 0; - f.cache().publish(scores, () => ++checks < 2); + f.cache().publish(panel, () => ++checks < 2); expect(checks).toBe(2); expect(f.entries()).toHaveLength(0); let now = 0; const h = actualCallback(f, { budget: 20, allowance: 5, clock: () => now, @@ -373,7 +380,7 @@ test('the actual workflow callback preserves the complete public API body; cance try { const h = actualCallback(f, { judge: (prompt, model, options) => callJudge(prompt, model, options) }); await h.run(h.options); - expect(create).toHaveBeenCalledTimes(1); + expect(create).toHaveBeenCalledTimes(SAMPLES); expect(create.mock.calls[0]).toEqual([{ model: resolveEvalModel('judge'), max_tokens: 8192, messages: [{ role: 'user', content: f.opts.prompt }], @@ -412,40 +419,40 @@ test('Ship sends its authorized 64k cap and compact response contract through th f.opts.structuredResponse = true; f.opts.maxTokens = 65_536; f.opts.stream = true; - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); } finally { stream.mockRestore(); } }); test('changing response serialization misses the cache even when prompt and model match', () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); f.opts.structuredResponse = true; expect(f.cache().lookup()).toBeNull(); - f.cache().publish(scores); + f.cache().publish(panel); expect(f.entries()).toHaveLength(2); - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); const description = WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description; try { WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description += ' Changed response contract.'; expect(f.cache().lookup()).toBeNull(); } finally { WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description = description; } - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); f.opts.structuredResponse = false; - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); }); test('the actual cap and streaming transport independently affect workflow cache identity', () => { - const f = fixture(); f.cache().publish(scores); + const f = fixture(); f.cache().publish(panel); f.opts.maxTokens = 65_536; expect(f.cache().lookup()).toBeNull(); - f.cache().publish(scores); + f.cache().publish(panel); f.opts.stream = true; expect(f.cache().lookup()).toBeNull(); - f.cache().publish(scores); + f.cache().publish(panel); expect(f.entries()).toHaveLength(3); - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); delete f.opts.maxTokens; delete f.opts.stream; - expect(f.cache().lookup()?.scores).toEqual(scores); + expect(f.cache().lookup()?.samples).toEqual(panel); }); test('the structured callback rejects incomplete, schema-invalid and below-threshold answers without cache credit', async () => { @@ -473,3 +480,96 @@ test('the structured callback rejects incomplete, schema-invalid and below-thres expect(validWorkflowJudgeScore({ ...scores, reasoning: Array(149).fill('word').join(' ') }, { clarity: 1, completeness: 1, actionability: 1 }, true)).toBe(true); } finally { stream.mockRestore(); diagnostics.mockRestore(); } }); + +// --- Judge panel policy (EVAL_POLICY.judge): fixed concurrent samples, per-dimension +// mean and boolean majority against unchanged thresholds, an erroring sample fails +// the whole panel and is never resampled. The provider is always a stub. +const panelScore = (clarity: number, completeness = 4, actionability = 4) => ({ clarity, completeness, actionability, reasoning: `c${clarity}` }); +const panelThresholds = { clarity: 3, completeness: 3, actionability: 4 }; +const panelRefusal = () => new JudgeRefusalError({ id: 'msg_1', _request_id: 'req_1', model: 'm', usage: { input_tokens: 1, output_tokens: 0 }, content: [] }); + +describe('judge panel', () => { + test('the pre-registered panel is three samples, and the helper restates EVAL_POLICY exactly', () => { + expect(EVAL_POLICY.judge.samples).toBe(3); + expect(JUDGE_PANEL_SAMPLES).toBe(EVAL_POLICY.judge.samples); + }); + + test('draws every sample concurrently before any resolves', async () => { + let started = 0; + const releases: Array<() => void> = []; + const panel = judgePanel(() => new Promise(resolve => { started += 1; releases.push(() => resolve(started)); })); + await Promise.resolve(); + expect(started).toBe(SAMPLES); + releases.forEach(release => release()); + expect(await panel).toHaveLength(SAMPLES); + }); + + test('an erroring sample fails the panel and is never resampled', async () => { + let calls = 0; + const panel = judgePanel(async () => { + calls += 1; + if (calls === 2) throw new Error('Judge returned non-JSON: nope'); + return panelScore(5); + }); + await expect(panel).rejects.toThrow('non-JSON'); + expect(calls).toBe(SAMPLES); + }); + + test('a refusal on every sample stays a provider refusal; a partial refusal is an ordinary failure', async () => { + await expect(judgePanel(async () => { throw panelRefusal(); })).rejects.toBeInstanceOf(JudgeRefusalError); + let calls = 0; + const partial = judgePanel(async () => { if (++calls === 1) throw panelRefusal(); return panelScore(4); }); + const error = await partial.then(() => null, (reason: unknown) => reason); + expect(error).toBeInstanceOf(Error); + expect(error).not.toBeInstanceOf(JudgeRefusalError); + expect(String(error)).toContain(`sample 1 of ${SAMPLES} failed beside scored samples`); + }); + + test('numeric dimensions gate on the per-dimension mean; one low sample can be outvoted, a low mean cannot', () => { + const outvoted = judgePanelMean([panelScore(2), panelScore(4), panelScore(4)], JUDGE_SCORE_DIMENSIONS); + expect(outvoted.clarity).toBeCloseTo(10 / 3); + expect(outvoted.clarity).toBeGreaterThanOrEqual(panelThresholds.clarity); + const low = judgePanelMean([panelScore(2), panelScore(2), panelScore(4)], JUDGE_SCORE_DIMENSIONS); + expect(low.clarity).toBeLessThan(panelThresholds.clarity); + // No compensation across dimensions: each is averaged on its own. + expect(judgePanelMean([panelScore(5, 1), panelScore(5, 1), panelScore(5, 1)], JUDGE_SCORE_DIMENSIONS).completeness).toBe(1); + }); + + test('malformed sample fields fail the panel instead of averaging to NaN', () => { + expect(() => judgePanelMean([panelScore(4), { ...panelScore(4), clarity: '4' as unknown as number }, panelScore(4)], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2 has non-numeric clarity'); + expect(() => judgePanelMean([panelScore(4), null as unknown as ReturnType], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2'); + expect(() => judgePanelMean([], JUDGE_SCORE_DIMENSIONS)).toThrow('no samples'); + }); + + test('boolean fields gate on a strict majority', () => { + const vote = (...values: boolean[]) => judgePanelMajority(values.map(value => ({ ok: value })), 'ok'); + expect(vote(true, true, false)).toBe(true); + expect(vote(true, false, false)).toBe(false); + expect(vote(true, false)).toBe(false); + expect(() => judgePanelMajority([{ ok: true }, { ok: 'yes' }], 'ok')).toThrow('sample 2 has non-boolean ok'); + }); + + test('reasoning keeps every sample, numbered, even for malformed samples', () => { + expect(judgePanelReasoning([panelScore(4), null, { reasoning: 7 }])).toBe('[sample 1] c4\n[sample 2] \n[sample 3] '); + }); + + test('the cache stores and validates only a complete panel against the mean', () => { + expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(true); + expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(2), panelScore(4)] }, panelThresholds)).toBe(false); + expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false); + expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false); + expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), { ...panelScore(4), clarity: 6 }] }, panelThresholds)).toBe(false); + expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4)], prompt: 'x' }, panelThresholds)).toBe(false); + expect(validWorkflowJudgePanel(panelScore(4), panelThresholds)).toBe(false); + }); + + test('every judge in the quality file samples through the panel, never a lone call', () => { + const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8'); + const calls = [...source.matchAll(/\b(?:callJudge<[^>(]*(?:<[^>]*>[^>(]*)*>|judge)\(/g)]; + expect(calls.length).toBeGreaterThanOrEqual(8); + for (const call of calls) { + expect(source.slice(Math.max(0, call.index! - 25), call.index), `unpaneled judge call at offset ${call.index}`).toMatch(/judgePanel\(\(\) => $/); + } + expect(source).not.toMatch(/\bscores\.reasoning\b|\bresult\.reasoning\b/); + }); +});