test(llm-judge): sample every judge as a pre-registered 3-sample panel

Each of the 24 skill-llm-eval judges now draws EVAL_POLICY.judge.samples
independent samples of the same prompt concurrently inside the unchanged
JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
against the unchanged threshold; booleans (would_browse, consistent) on a
strict majority. An erroring sample fails the whole panel and is never
resampled; a refusal is an unscored panel only when every sample refused.
callJudge's 429 backoff stays: it is transport before any model output.

The workflow-judge cache stores and validates only complete panels, and its
identity now records the panel and zero file retries. Harness tests that
pinned one provider call per case now pin the panel size.
This commit is contained in:
garrytan committed 2026-09-29 19:11:04 +00:00
1 parent adced7e046
commit 3f68572cab
7 files changed
+298 -112

No files matched your search

+7 -7
View File
@@ -11,7 +11,7 @@ import { selectTests } from './helpers/test-selection';
import { E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data';
import { selectPrProfile } from '../scripts/test-pr-profile';
import { JUDGE_MS } from './helpers/eval-budgets';
import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge';
import { COOKIE_MANUAL_REVIEW_FILE, getCookieWorkflowManualReview, isManualReviewEntry } from './helpers/cookie-workflow-manual-review';
const ROOT = resolve(import.meta.dir, '..');
@@ -70,7 +70,7 @@ function actualCookieCallback(root: string, overrides: {
const records: EvalTestEntry[] = [];
const attempts = new Map<string, { attempt: number }>();
let callback: () => Promise<void> = async () => { throw new Error('Judge callback was not registered'); };
new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', registration)(
new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', 'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS', registration)(
(_suite: string, names: string[], run: () => void) => { expect(names).toEqual([NAME]); run(); },
(name: string, run: () => Promise<void>, budget: number) => { expect(name).toBe(NAME); expect(budget).toBe(JUDGE_MS + 10_000); callback = run; },
root, buildCookieWorkflowJudgeInput, (_kind: string, explicit?: string) => explicit ?? 'fixture-model',
@@ -85,7 +85,7 @@ function actualCookieCallback(root: string, overrides: {
attempts, overrides.clock ? { now: overrides.clock } : performance,
overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout,
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS,
);
return { run: () => callback(), requests, records, attempts };
}
@@ -96,7 +96,7 @@ describe('cookie workflow judge input', () => {
approveFixture(root);
const h = actualCookieCallback(root, { judge: async () => { throw refusal(); } });
await h.run();
expect(h.requests).toHaveLength(1);
expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
expect(h.records).toHaveLength(1);
expect(h.records[0]).toMatchObject({ passed: false, execution: 'executed', exit_reason: 'provider_refusal' });
expect(isManualReviewEntry(h.records[0])).toBe(true);
@@ -161,7 +161,7 @@ describe('cookie workflow judge input', () => {
const root = fixture(); approveFixture(root);
let calls = 0;
const h = actualCookieCallback(root, { judge: async () => {
if (++calls === 1) return { ...passingScore, clarity: 1 };
if (++calls <= JUDGE_PANEL_SAMPLES) return { ...passingScore, clarity: 1 };
throw refusal();
} });
await expect(h.run()).rejects.toThrow();
@@ -275,7 +275,7 @@ describe('cookie workflow judge input', () => {
let scores = passingScore;
const h = actualCookieCallback(root, { judge: async () => scores });
await h.run();
expect(h.requests).toHaveLength(1);
expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
expect(h.requests[0].prompt).toBe(input.prompt);
expect(h.requests[0].model).toBe(COOKIE_WORKFLOW_JUDGE.model);
expect(h.requests[0].signal).toBeInstanceOf(AbortSignal);
@@ -283,7 +283,7 @@ describe('cookie workflow judge input', () => {
expect(existsSync(join(root, 'cache'))).toBe(false);
const fresh = actualCookieCallback(root);
await fresh.run();
expect(fresh.requests).toHaveLength(1);
expect(fresh.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
for (const dimension of ['clarity', 'completeness', 'actionability'] as const) {
scores = { ...COOKIE_WORKFLOW_JUDGE.thresholds, [dimension]: COOKIE_WORKFLOW_JUDGE.thresholds[dimension] - 1, reasoning: 'Synthetic failing fixture score' };
await expect(h.run()).rejects.toThrow();
+59
View File
@@ -23,6 +23,8 @@ export interface JudgeScore {
reasoning: string;
}
export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const;
export interface JudgeRefusalEvidence {
stop_reason: 'refusal';
response_id: string | null;
@@ -196,6 +198,63 @@ export async function callJudge<T>(
}
}
/**
* Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this
* helper (imported by many paid tests) does not pull the quarantine registry
* into their touchfile closure. test/judge-panel.test.ts pins the two equal.
*/
export const JUDGE_PANEL_SAMPLES = 3;
/**
* Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of
* independent samples of the SAME prompt concurrently, inside its unchanged
* JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
* against the unchanged minimum; boolean fields gate on a strict majority.
* A sample that errors (refusal, truncation, non-JSON, malformed field) fails
* the whole panel and is never resampled. callJudge's 429 backoff happens
* before any model output exists, so it is transport, not a verdict retry.
*/
export async function judgePanel<T>(sample: () => Promise<T>): Promise<T[]> {
const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample()));
const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []);
if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult<T>).value);
const first = failures[0]!;
// A refusal is an unscored panel only when EVERY sample refused; a partial
// refusal beside scored samples is an ordinary failed panel.
if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) {
throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`);
}
throw first.reason;
}
/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */
export function judgePanelMean<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, keys: readonly K[]): Record<K, number> {
if (samples.length === 0) throw new Error('Judge panel has no samples');
return Object.fromEntries(keys.map(key => {
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value));
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`);
return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length];
})) as Record<K, number>;
}
/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */
export function judgePanelMajority<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, key: K): boolean {
if (samples.length === 0) throw new Error('Judge panel has no samples');
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
const bad = values.findIndex(value => typeof value !== 'boolean');
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`);
return values.filter(value => value === true).length * 2 > values.length;
}
/** Sample reasoning lines, numbered, for the collector record. */
export function judgePanelReasoning(samples: ReadonlyArray<unknown>): string {
return samples.map((sample, index) => {
const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined;
return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`;
}).join('\n');
}
/**
* Score documentation quality on clarity/completeness/actionability (1-5).
*/
+7 -1
View File
@@ -72,7 +72,12 @@ export const CASE_CI_EXCLUDE: Record<string, { reason: string; tracking: string
* new-policy trials; exit at >= `exit.rate` over >=
* `exit.minTrials`; at most `capFraction` of each tier's
* blocking cases; an entry expires after `expiryWeeklyRuns`.
* drift - one-sided Fisher exact alarm between input-identity series.
* judge - a judge case draws `samples` independent samples of one
* prompt concurrently; numeric dimensions gate on the panel
* mean against the unchanged threshold, booleans on a strict
* majority; an erroring sample fails the panel, never resampled.
* drift - one-sided Fisher exact alarm between input-identity series
* (Holm-controlled across the cases tested in one report).
* infraRedispatch - a census whose every red verdict is machine-classified
* INFRA or INCOMPLETE may be re-dispatched this many times as
* a new run; both runs are reported.
@@ -86,6 +91,7 @@ export const EVAL_POLICY = {
capFraction: 0.10,
expiryWeeklyRuns: 8,
},
judge: { samples: 3 },
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
infraRedispatch: 1,
} as const;
+24 -11
View File
@@ -4,7 +4,7 @@ import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
import { JUDGE_MS } from './eval-budgets';
import type { JudgeScore } from './llm-judge';
import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input';
import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache,
type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache';
@@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres
|| typeof value.reasoning !== 'string'
|| (structuredResponse && (!value.reasoning.trim()
|| value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false;
return (['clarity', 'completeness', 'actionability'] as const).every(key =>
return JUDGE_SCORE_DIMENSIONS.every(key =>
typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5);
}
const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 };
/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */
export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array<JudgeScore & EvalCacheValue> } {
if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples'
|| !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES
|| !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false;
const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS);
return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]);
}
export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null;
lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null;
/** The attempt guard is rechecked after synchronous input/provenance reads. */
publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined;
publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined;
} {
const env = opts.env ?? process.env;
const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined };
const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined };
const pr = Number(env.EVALS_CACHE_PR);
// Runtime ID is the immutable CI image manifest, not a mutable image tag.
// Nonstandard Node/Bun preload code or custom model endpoints need a separate
@@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
prompts: { [opts.testName]: prompt },
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1,
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0,
panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' },
...(opts.stream ? { stream: true } : {}),
...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) },
@@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
return {
lookup() {
const result = lookupEvalInputCache({ ...common, identity: before,
validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) });
validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) });
return result.status === 'reused'
? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null;
? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null;
},
publish(scores, isActive = () => true) {
publish(samples, isActive = () => true) {
// Caller reaches here ONLY after its actual assertions passed. A later
// failed case in the file does not erase this independently completed case.
if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) };
if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
const after = currentIdentity();
const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID;
if (!after || !runId || !isActive()) return;
@@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
cancelled: false, skipped: 0, failed: 0, passed: 1,
cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }],
source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() },
result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning },
result: panel,
} });
// A slow synchronous write can consume the recording allowance. The
// caller withdraws this new receipt if its final deadline check fails.
+55 -46
View File
@@ -14,7 +14,7 @@ import { afterAll, expect } from 'bun:test';
import { JUDGE_MS } from './helpers/eval-budgets';
import * as fs from 'fs';
import * as path from 'path';
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelMajority, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS } from './helpers/llm-judge';
import { ENG_REVIEW_EXCERPT } from './helpers/workflow-excerpt';
import type { JudgeScore } from './helpers/llm-judge';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA, type WorkflowJudgeInput } from './helpers/workflow-judge-input';
@@ -100,8 +100,9 @@ describeIfSelected('LLM-as-judge quality evals', [
// rewrites the pin).
const section = sliceBrowseSection('## Snapshot Flags');
const scores = await judge('browse skill reference (flags + commands)', section);
console.log('Browse SKILL.md scores:', JSON.stringify(scores, null, 2));
const samples = await judgePanel(() => judge('browse skill reference (flags + commands)', section));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('Browse SKILL.md panel:', JSON.stringify({ mean: scores, samples }, null, 2));
const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json');
const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8'));
@@ -120,9 +121,9 @@ describeIfSelected('LLM-as-judge quality evals', [
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4 && regressions.length === 0,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: regressions.length ? `${scores.reasoning} | ${regressions.join('; ')}` : scores.reasoning,
judge_reasoning: regressions.length ? `${judgePanelReasoning(samples)} | ${regressions.join('; ')}` : judgePanelReasoning(samples),
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
@@ -144,8 +145,9 @@ describeIfSelected('LLM-as-judge quality evals', [
if (setupStart < 0 || setupEnd < 0) throw new Error('browse/SKILL.md: setup block not found — regenerate with: bun run gen:skill-docs');
const section = content.slice(setupStart, setupEnd);
const scores = await judge('setup/binary discovery instructions', section);
console.log('Setup block scores:', JSON.stringify(scores, null, 2));
const samples = await judgePanel(() => judge('setup/binary discovery instructions', section));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('Setup block panel:', JSON.stringify({ mean: scores, samples }, null, 2));
evalCollector?.addTest({
name: 'setup block',
@@ -153,9 +155,9 @@ describeIfSelected('LLM-as-judge quality evals', [
tier: 'llm-judge',
passed: scores.actionability >= 3 && scores.clarity >= 3,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
// Setup block is intentionally minimal (binary discovery only).
@@ -203,7 +205,7 @@ describeIfSelected('QA skill quality evals', ['qa/SKILL.md workflow', 'qa/SKILL.
startMarker: '# /qa: Test', endMarker: null,
references: ['qa/templates/functional-report-template.md'] }).text;
const scores = await callJudge<JudgeScore>(`You are evaluating the quality of a QA testing workflow document for an AI coding agent.
const samples = await judgePanel(() => callJudge<JudgeScore>(`You are evaluating the quality of a QA testing workflow document for an AI coding agent.
The agent reads this source-file bundle to select browser, native functional or mixed
surfaces, explore with bounded probes, reproduce and diagnose defects, add a regression
@@ -222,8 +224,9 @@ Respond with ONLY valid JSON:
Here is the QA workflow to evaluate:
${section}`);
console.log('QA workflow scores:', JSON.stringify(scores, null, 2));
${section}`));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('QA workflow panel:', JSON.stringify({ mean: scores, samples }, null, 2));
evalCollector?.addTest({
name: 'qa/SKILL.md workflow',
@@ -231,9 +234,9 @@ ${section}`);
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
@@ -247,7 +250,7 @@ ${section}`);
const t0 = Date.now();
const section = sliceQaPatterns('## Health Score Rubric');
const scores = await callJudge<JudgeScore>(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score.
const samples = await judgePanel(() => callJudge<JudgeScore>(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score.
The agent uses this rubric after QA testing a website. It needs to:
1. Understand each scoring category and what counts as a deduction
@@ -264,8 +267,9 @@ Respond with ONLY valid JSON:
Here is the rubric to evaluate:
${section}`);
console.log('QA health rubric scores:', JSON.stringify(scores, null, 2));
${section}`));
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log('QA health rubric panel:', JSON.stringify({ mean: scores, samples }, null, 2));
evalCollector?.addTest({
name: 'qa/SKILL.md health rubric',
@@ -273,9 +277,9 @@ ${section}`);
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
@@ -294,7 +298,7 @@ ${section}`);
const diffAwareSection = sliceQaPatterns('### Diff-aware', '### Full');
const rulesSection = sliceQaPatterns('## Important Rules');
const result = await callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario.
const samples = await judgePanel(() => callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario.
SCENARIO:
A user runs /qa (a browser-based QA testing skill). The branch diff shows ONLY prompt template files and config file changes — no routes, views, controllers, components, or CSS were changed. The changes are "purely backend" with no obvious UI surface.
@@ -318,9 +322,10 @@ Respond with ONLY valid JSON:
Rules:
- would_browse should be true if the document instructs the agent to always use the browser regardless of diff content
- would_browse should be false if the document allows the agent to skip browser testing for non-UI changes
- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`);
- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`));
const result = { would_browse: judgePanelMajority(samples, 'would_browse'), ...judgePanelMean(samples, ['confidence'] as const) };
console.log('QA anti-refusal result:', JSON.stringify(result, null, 2));
console.log('QA anti-refusal panel:', JSON.stringify({ result, samples }, null, 2));
evalCollector?.addTest({
name: 'qa/SKILL.md anti-refusal',
@@ -328,9 +333,9 @@ Rules:
tier: 'llm-judge',
passed: result.would_browse === true && result.confidence >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { would_browse: result.would_browse ? 1 : 0, confidence: result.confidence },
judge_reasoning: result.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(result.would_browse).toBe(true);
@@ -362,7 +367,7 @@ describeIfSelected('Cross-skill consistency evals', ['cross-skill greptile consi
extractGrepLines(retroContent, 'retro/SKILL.md'),
].join('\n\n');
const result = await callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently.
const samples = await judgePanel(() => callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently.
INTENDED ARCHITECTURE:
- greptile-history has TWO paths: per-project (~/.gstack/projects/{slug}/greptile-history.md) and global (~/.gstack/greptile-history.md)
@@ -383,9 +388,10 @@ Evaluate consistency. Respond with ONLY valid JSON:
"reasoning": "brief explanation"
}
score (1-5): 5 = perfectly consistent, 1 = contradictory`);
score (1-5): 5 = perfectly consistent, 1 = contradictory`));
const result = { consistent: judgePanelMajority(samples, 'consistent'), ...judgePanelMean(samples, ['score'] as const) };
console.log('Cross-skill consistency:', JSON.stringify(result, null, 2));
console.log('Cross-skill consistency panel:', JSON.stringify({ result, samples }, null, 2));
evalCollector?.addTest({
name: 'cross-skill greptile consistency',
@@ -393,9 +399,9 @@ score (1-5): 5 = perfectly consistent, 1 = contradictory`);
tier: 'llm-judge',
passed: result.consistent && result.score >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: { consistency_score: result.score },
judge_reasoning: result.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(result.consistent).toBe(true);
@@ -439,7 +445,8 @@ async function runWorkflowJudge(opts: {
const workDeadline = started + JUDGE_MS;
let stage: 'input' | 'judge' | 'validation' | 'recording' = 'input';
let finalized = false;
let scores: JudgeScore | undefined;
let samples: JudgeScore[] | undefined;
let scores: Record<typeof JUDGE_SCORE_DIMENSIONS[number], number> | undefined;
let manualReview: ManualJudgeReview | undefined;
let customInputMetadata: { prompt: string; model: string } | undefined;
let reused: ReturnType<ReturnType<typeof prepareWorkflowJudgeCache>['lookup']> = null;
@@ -458,19 +465,19 @@ async function runWorkflowJudge(opts: {
evalCollector?.addTest({
name: opts.testName, suite: opts.suite, tier: 'llm-judge', passed, attempt,
duration_ms: Math.max(0, performance.now() - started),
cost_usd: reused || !scores ? 0 : 0.02,
cost_usd: reused || !samples ? 0 : 0.02 * samples.length,
execution: reused ? 'reused' : 'executed',
...customInputMetadata,
...(manualReview ? { manual_review: manualReview } : {}),
...(reused ? { reused_from: { input_key: reused.reuse.key, run_id: reused.reuse.source.runId,
revision: reused.reuse.source.revision, completed_at: new Date(reused.reuse.source.completedAt).toISOString() } } : {}),
...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning } : {}),
...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability } } : {}),
...(samples ? { judge_reasoning: judgePanelReasoning(samples) } : {}),
...(passed ? {} : { exit_reason: error instanceof JudgeRefusalError ? 'provider_refusal'
: error instanceof Error && error.name === 'WorkflowJudgeDeadline' ? 'timeout'
: error instanceof Error && error.name === 'WorkflowJudgeSuperseded' ? 'cancelled'
: stage === 'validation' ? 'validation_failed' : 'harness_error',
error: `${error instanceof Error ? error.message : String(error)}${scores ? '' : error instanceof JudgeRefusalError
error: `${error instanceof Error ? error.message : String(error)}${samples ? '' : error instanceof JudgeRefusalError
? '\nNo automated score; provider refusal usage retained when manually accepted; cost unavailable.'
: '\nNo completed model response; cost and usage unavailable.'}` }),
});
@@ -508,11 +515,11 @@ async function runWorkflowJudge(opts: {
checkActive();
stage = 'judge';
const maxTokens = opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS;
let result: JudgeScore;
let result: JudgeScore[];
try {
result = reused?.scores ?? await callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,
result = reused?.samples ?? await judgePanel(() => callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,
...(opts.stream ? { stream: true } : {}),
...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) });
...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) }));
} catch (error) {
checkActive();
if (error instanceof JudgeRefusalError && customInputMetadata) {
@@ -529,20 +536,21 @@ async function runWorkflowJudge(opts: {
throw error;
}
checkActive();
scores = result;
samples = result;
console.log(`[workflow-judge] ${opts.testName}: ${reused ? `reused ${reused.reuse.source.runId} @ ${reused.reuse.source.revision} (${new Date(reused.reuse.source.completedAt).toISOString()})` : 'executed'}`);
console.log(`${opts.testName} scores:`, JSON.stringify(scores, null, 2));
stage = 'validation';
if (opts.structuredResponse && !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true)) {
if (opts.structuredResponse && !samples.every(sample => validWorkflowJudgeScore(sample as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true))) {
throw new Error('Structured workflow judge violated the response schema');
}
scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
console.log(`${opts.testName} panel:`, JSON.stringify({ mean: scores, samples }, null, 2));
expect(scores.clarity).toBeGreaterThanOrEqual(thresholds.clarity);
expect(scores.completeness).toBeGreaterThanOrEqual(thresholds.completeness);
expect(scores.actionability).toBeGreaterThanOrEqual(thresholds.actionability);
checkActive();
stage = 'recording';
arm();
const discardReceipt = reused ? undefined : cache.publish(scores, active);
const discardReceipt = reused ? undefined : cache.publish(samples, active);
try { checkActive(); finish(true); }
catch (error) { discardReceipt?.(); throw error; }
};
@@ -792,7 +800,7 @@ describeIfSelected('Voice directive eval', ['voice directive tone'], () => {
const voiceEnd = content.indexOf('\n## ', voiceStart + 1);
const voiceSection = content.slice(voiceStart, voiceEnd > 0 ? voiceEnd : voiceStart + 3000);
const result = await callJudge<{
const samples = await judgePanel(() => callJudge<{
directness: number;
concreteness: number;
avoids_corporate: number;
@@ -812,9 +820,10 @@ Return JSON only:
{"directness": N, "concreteness": N, "avoids_corporate": N, "avoids_ai_vocabulary": N, "connects_user_outcomes": N, "reasoning": "..."}
THE VOICE DIRECTIVE:
${voiceSection}`);
${voiceSection}`));
const result = judgePanelMean(samples, ['directness', 'concreteness', 'avoids_corporate', 'avoids_ai_vocabulary', 'connects_user_outcomes'] as const);
console.log('Voice directive scores:', JSON.stringify(result, null, 2));
console.log('Voice directive panel:', JSON.stringify({ mean: result, samples }, null, 2));
evalCollector?.addTest({
name: 'voice directive tone',
@@ -823,7 +832,7 @@ ${voiceSection}`);
passed: result.directness >= 4 && result.concreteness >= 4 && result.avoids_corporate >= 4
&& result.avoids_ai_vocabulary >= 4 && result.connects_user_outcomes >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
cost_usd: 0.02 * samples.length,
judge_scores: {
directness: result.directness,
concreteness: result.concreteness,
@@ -831,7 +840,7 @@ ${voiceSection}`);
avoids_ai_vocabulary: result.avoids_ai_vocabulary,
connects_user_outcomes: result.connects_user_outcomes,
},
judge_reasoning: result.reasoning,
judge_reasoning: judgePanelReasoning(samples),
});
expect(result.directness).toBeGreaterThanOrEqual(4);
+146 -46
View File
@@ -1,18 +1,22 @@
import { afterEach, expect, spyOn, test } from 'bun:test';
import { afterEach, describe, expect, spyOn, test } from 'bun:test';
import { Messages } from '@anthropic-ai/sdk/resources/messages';
import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMajority, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge';
import { EVAL_POLICY } from './helpers/periodic-exclude-data';
import { getCookieWorkflowManualReview } from './helpers/cookie-workflow-manual-review';
import { resolveEvalModel } from '../lib/eval-model';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { execFileSync } from 'node:child_process';
import { prepareWorkflowJudgeCache, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
import { prepareWorkflowJudgeCache, validWorkflowJudgePanel, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA } from './helpers/workflow-judge-input';
const roots: string[] = [];
afterEach(() => { for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); });
const scores = { clarity: 4, completeness: 5, actionability: 4, reasoning: 'Concrete steps' };
const SAMPLES = JUDGE_PANEL_SAMPLES;
const panelOf = (sample: typeof scores) => Array.from({ length: SAMPLES }, () => sample);
const panel = panelOf(scores);
function fixture() {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-judge-cache-')); roots.push(root);
const files = {
@@ -53,9 +57,9 @@ function fixture() {
}
test('the audited adapter reuses only the exact completed score and original provenance', () => {
const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(scores);
const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(panel);
expect(f.entries()).toHaveLength(1);
const reused = f.cache().lookup(); expect(reused?.scores).toEqual(scores);
const reused = f.cache().lookup(); expect(reused?.samples).toEqual(panel);
expect(reused?.reuse.source.runId).toBe('free-cache-test');
expect(reused?.reuse.source.revision).toMatch(/^[a-f0-9]{40}$/);
expect(reused?.reuse.source.completedAt).toBeLessThanOrEqual(Date.now());
@@ -73,11 +77,11 @@ test('the dependency closure includes actual installed SDK bytes and local trans
});
test('release-label changes preserve reuse; other package semantics invalidate it', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
const file = path.join(f.root, 'package.json');
const original = JSON.parse(fs.readFileSync(file, 'utf8'));
fs.writeFileSync(file, JSON.stringify({ ...original, version: '2.0.0' }, null, 2));
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
for (const change of [{ scripts: { 'test:gate': 'changed command' } }, { dependencies: { 'some-sdk': '2.0.0' } }]) {
fs.writeFileSync(file, JSON.stringify({ ...original, ...change, version: '2.0.0' }));
expect(f.cache().lookup()).toBeNull();
@@ -89,7 +93,7 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in
'scripts/test-paid-shards.ts', 'scripts/test-strict-output.ts', 'scripts/eval-select.ts',
'scripts/test-pr-profile.ts', '.github/workflows/evals.yml']) {
test(`changes in ${file} require new evaluation`, () => {
const f = fixture(); f.cache().publish(scores); const target = path.join(f.root, file);
const f = fixture(); f.cache().publish(panel); const target = path.join(f.root, file);
fs.appendFileSync(target, file.endsWith('.json') ? ' ' : '\n// changed');
f.refreshPrompt(); expect(f.cache().lookup()).toBeNull();
});
@@ -98,8 +102,8 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in
test('changed sources during an attempt and mismatched actual prompt cannot publish', () => {
const f = fixture(); const before = f.cache();
fs.appendFileSync(path.join(f.root, 'example/sections/review.md'), 'new finding');
before.publish(scores); expect(f.entries()).toHaveLength(0);
f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(scores);
before.publish(panel); expect(f.entries()).toHaveLength(0);
f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(panel);
expect(f.entries()).toHaveLength(0);
});
@@ -108,52 +112,52 @@ for (const [key, value] of Object.entries({ EVALS_FRESH: '1', EVALS_TIER: 'perio
EVALS_CACHE_REPOSITORY: '', NODE_OPTIONS: '--require=unknown', BUN_OPTIONS: '--preload=unknown',
ANTHROPIC_BASE_URL: 'https://custom-provider.example.test' })) {
test(`${key}=${value} is fresh or ineligible`, () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.env = { ...f.env, [key]: value }; const cache = f.cache();
expect(cache.lookup()).toBeNull(); cache.publish(scores); expect(f.entries()).toHaveLength(1);
expect(cache.lookup()).toBeNull(); cache.publish(panel); expect(f.entries()).toHaveLength(1);
});
}
test('runtime/model/threshold changes miss, and retries never reuse or publish', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
for (const overrides of [{ GSTACK_EVAL_MODEL_JUDGE: 'different-model' }, { EVALS_CACHE_RUNTIME_ID: 'c'.repeat(64) }]) {
f.opts.env = { ...f.env, ...overrides }; expect(f.cache().lookup()).toBeNull();
}
f.opts.env = f.env; f.opts.thresholds.clarity = 5; expect(f.cache().lookup()).toBeNull();
f.opts.thresholds.clarity = 4; f.opts.attempt = 2; const retry = f.cache();
expect(retry.lookup()).toBeNull(); retry.publish(scores); expect(f.entries()).toHaveLength(1);
expect(retry.lookup()).toBeNull(); retry.publish(panel); expect(f.entries()).toHaveLength(1);
});
test('frontier reader calibration cannot reuse a score from the unspecified-reader rubric', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
const original = f.opts.prompt;
f.opts.agentCapability = 'frontier'; f.refreshPrompt();
expect(f.opts.prompt).not.toBe(original);
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(2);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
delete f.opts.agentCapability; f.refreshPrompt();
expect(f.opts.prompt).toBe(original);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('a pinned workflow judge model overrides the global model and changes the cache identity', () => {
const f = fixture();
f.opts.model = 'claude-sonnet-4-6';
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(1);
f.opts.env = { ...f.env, GSTACK_EVAL_MODEL_JUDGE: 'different-global-model' };
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
f.opts.model = 'claude-opus-4-7';
expect(f.cache().lookup()).toBeNull();
});
test('failed assertions, missing provenance, and missing imported dependencies cannot supply a receipt', () => {
const f = fixture(); f.cache().publish({ ...scores, clarity: 3 }); expect(f.entries()).toHaveLength(0);
f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(scores); expect(f.entries()).toHaveLength(0);
const f = fixture(); f.cache().publish(panelOf({ ...scores, clarity: 3 })); expect(f.entries()).toHaveLength(0);
f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(panel); expect(f.entries()).toHaveLength(0);
f.opts.env = f.env; fs.unlinkSync(path.join(f.root, 'test/helpers/nested.ts'));
f.cache().publish(scores); expect(f.entries()).toHaveLength(0);
f.cache().publish(panel); expect(f.entries()).toHaveLength(0);
});
test('cached payload schema remains small and cannot carry operational fields', () => {
@@ -167,8 +171,9 @@ test('workflow registration preserves model work and reserves only terminal-reco
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const body = source.split('async function runWorkflowJudge')[1]!.split('// Block 1:')[0]!;
const stages = ['workflowJudgeAttempts.set', 'readWorkflowJudgeInput(', 'cache.lookup()',
'callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(scores, active)']
'judgePanel(() => callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
'scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);',
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(samples, active)']
.map(stage => body.indexOf(stage));
expect(stages.every(position => position >= 0)).toBe(true);
expect(stages).toEqual([...stages].sort((a, b) => a - b));
@@ -203,6 +208,7 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
'evalCollector', 'expect', 'console', 'performance', 'JUDGE_MS', 'WORKFLOW_JUDGE_RECORD_MS',
'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'resolveEvalModel',
'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore',
'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS',
`${javascript}\nreturn runWorkflowJudge;`)(
f.root, overrides.read ?? readWorkflowJudgeInput, buildWorkflowJudgePrompt,
(options: WorkflowCacheOptions) => (overrides.prepare ?? prepareWorkflowJudgeCache)({ ...options, env: f.env }),
@@ -213,7 +219,8 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
overrides.clock ? { now: overrides.clock } : performance, overrides.budget ?? 120_000, overrides.allowance ?? 5_000,
overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout,
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore);
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore,
judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS);
return { run, records, signals, prompts, attempts, options: { ...f.opts, suite: 'Cache regression' } };
}
@@ -223,7 +230,7 @@ test('the actual workflow callback preserves the pinned model and frontier rubri
const actual = actualCallback(f, { judge: async (_prompt, model) => { models.push(model); return scores; } });
await actual.run({ ...actual.options, model: 'claude-sonnet-4-6', agentCapability: 'frontier',
readInput: () => readWorkflowJudgeInput(f.opts) });
expect(models).toEqual(['claude-sonnet-4-6']);
expect(models).toEqual(Array(SAMPLES).fill('claude-sonnet-4-6'));
expect(actual.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
expect(actual.records[0]).toMatchObject({ passed: true, model: 'claude-sonnet-4-6', prompt: actual.prompts[0] });
});
@@ -239,7 +246,7 @@ test.each(['ship', 'review'])('the registered %s callback sends the frontier rub
endMarker: f.opts.endMarker, references: [] };
const passing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 3 }) });
await passing.run(options);
expect(passing.prompts).toHaveLength(1);
expect(passing.prompts).toHaveLength(SAMPLES);
expect(passing.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
expect(passing.records[0]).toMatchObject({ passed: true, execution: 'executed', judge_scores: { clarity: 3 } });
const failing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 2 }) });
@@ -254,8 +261,8 @@ test('the actual workflow callback executes once, reuses with provenance, and pr
const f = fixture(); const first = actualCallback(f);
const options = { ...f.opts, suite: 'Cache regression' };
await first.run(options);
expect(first.prompts).toEqual([f.opts.prompt]); expect(f.entries()).toHaveLength(1);
expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 });
expect(first.prompts).toEqual(Array(SAMPLES).fill(f.opts.prompt)); expect(f.entries()).toHaveLength(1);
expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 * SAMPLES });
expect(first.records[0]).not.toHaveProperty('prompt');
expect(first.records[0]).not.toHaveProperty('model');
const reused = actualCallback(f, { judge: async () => ({ ...scores, clarity: 1 }) });
@@ -312,13 +319,13 @@ test('a superseding attempt cancels its predecessor before either can record a s
});
test('a failed input read consumes attempt one and prevents a retry from borrowing or publishing a receipt', async () => {
const f = fixture(); f.cache().publish(scores); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8');
const f = fixture(); f.cache().publish(panel); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8');
let reads = 0;
const h = actualCallback(f, { read: options => { if (++reads === 1) throw new Error('Missing workflow fixture'); return readWorkflowJudgeInput(options); } });
await expect(h.run(h.options)).rejects.toThrow('Missing workflow fixture');
expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'harness_error' });
await h.run(h.options);
expect(h.prompts).toHaveLength(1);
expect(h.prompts).toHaveLength(SAMPLES);
expect(h.records.map(record => record.execution)).toEqual(['executed', 'executed']);
expect(h.attempts.get(f.opts.testName).attempt).toBe(2);
expect(fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8')).toBe(receipt);
@@ -333,14 +340,14 @@ test('monotonic expiry after a synchronous preparation or late model response re
await expect(h.run(h.options)).rejects.toThrow('deadline');
expect(h.records).toHaveLength(1);
expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'timeout', duration_ms: 21 });
expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : 1);
expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : SAMPLES);
expect(f.entries()).toHaveLength(0);
}
});
test('publication rechecks after input scanning and withdraws a receipt if recording expires', async () => {
const f = fixture(); let checks = 0;
f.cache().publish(scores, () => ++checks < 2);
f.cache().publish(panel, () => ++checks < 2);
expect(checks).toBe(2); expect(f.entries()).toHaveLength(0);
let now = 0;
const h = actualCallback(f, { budget: 20, allowance: 5, clock: () => now,
@@ -373,7 +380,7 @@ test('the actual workflow callback preserves the complete public API body; cance
try {
const h = actualCallback(f, { judge: (prompt, model, options) => callJudge<typeof scores>(prompt, model, options) });
await h.run(h.options);
expect(create).toHaveBeenCalledTimes(1);
expect(create).toHaveBeenCalledTimes(SAMPLES);
expect(create.mock.calls[0]).toEqual([{
model: resolveEvalModel('judge'), max_tokens: 8192,
messages: [{ role: 'user', content: f.opts.prompt }],
@@ -412,40 +419,40 @@ test('Ship sends its authorized 64k cap and compact response contract through th
f.opts.structuredResponse = true;
f.opts.maxTokens = 65_536;
f.opts.stream = true;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
} finally { stream.mockRestore(); }
});
test('changing response serialization misses the cache even when prompt and model match', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.structuredResponse = true;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(2);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
const description = WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description;
try {
WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description += ' Changed response contract.';
expect(f.cache().lookup()).toBeNull();
} finally { WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description = description; }
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
f.opts.structuredResponse = false;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('the actual cap and streaming transport independently affect workflow cache identity', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.maxTokens = 65_536;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
f.opts.stream = true;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(3);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
delete f.opts.maxTokens;
delete f.opts.stream;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('the structured callback rejects incomplete, schema-invalid and below-threshold answers without cache credit', async () => {
@@ -473,3 +480,96 @@ test('the structured callback rejects incomplete, schema-invalid and below-thres
expect(validWorkflowJudgeScore({ ...scores, reasoning: Array(149).fill('word').join(' ') }, { clarity: 1, completeness: 1, actionability: 1 }, true)).toBe(true);
} finally { stream.mockRestore(); diagnostics.mockRestore(); }
});
// --- Judge panel policy (EVAL_POLICY.judge): fixed concurrent samples, per-dimension
// mean and boolean majority against unchanged thresholds, an erroring sample fails
// the whole panel and is never resampled. The provider is always a stub.
const panelScore = (clarity: number, completeness = 4, actionability = 4) => ({ clarity, completeness, actionability, reasoning: `c${clarity}` });
const panelThresholds = { clarity: 3, completeness: 3, actionability: 4 };
const panelRefusal = () => new JudgeRefusalError({ id: 'msg_1', _request_id: 'req_1', model: 'm', usage: { input_tokens: 1, output_tokens: 0 }, content: [] });
describe('judge panel', () => {
test('the pre-registered panel is three samples, and the helper restates EVAL_POLICY exactly', () => {
expect(EVAL_POLICY.judge.samples).toBe(3);
expect(JUDGE_PANEL_SAMPLES).toBe(EVAL_POLICY.judge.samples);
});
test('draws every sample concurrently before any resolves', async () => {
let started = 0;
const releases: Array<() => void> = [];
const panel = judgePanel(() => new Promise<number>(resolve => { started += 1; releases.push(() => resolve(started)); }));
await Promise.resolve();
expect(started).toBe(SAMPLES);
releases.forEach(release => release());
expect(await panel).toHaveLength(SAMPLES);
});
test('an erroring sample fails the panel and is never resampled', async () => {
let calls = 0;
const panel = judgePanel(async () => {
calls += 1;
if (calls === 2) throw new Error('Judge returned non-JSON: nope');
return panelScore(5);
});
await expect(panel).rejects.toThrow('non-JSON');
expect(calls).toBe(SAMPLES);
});
test('a refusal on every sample stays a provider refusal; a partial refusal is an ordinary failure', async () => {
await expect(judgePanel(async () => { throw panelRefusal(); })).rejects.toBeInstanceOf(JudgeRefusalError);
let calls = 0;
const partial = judgePanel(async () => { if (++calls === 1) throw panelRefusal(); return panelScore(4); });
const error = await partial.then(() => null, (reason: unknown) => reason);
expect(error).toBeInstanceOf(Error);
expect(error).not.toBeInstanceOf(JudgeRefusalError);
expect(String(error)).toContain(`sample 1 of ${SAMPLES} failed beside scored samples`);
});
test('numeric dimensions gate on the per-dimension mean; one low sample can be outvoted, a low mean cannot', () => {
const outvoted = judgePanelMean([panelScore(2), panelScore(4), panelScore(4)], JUDGE_SCORE_DIMENSIONS);
expect(outvoted.clarity).toBeCloseTo(10 / 3);
expect(outvoted.clarity).toBeGreaterThanOrEqual(panelThresholds.clarity);
const low = judgePanelMean([panelScore(2), panelScore(2), panelScore(4)], JUDGE_SCORE_DIMENSIONS);
expect(low.clarity).toBeLessThan(panelThresholds.clarity);
// No compensation across dimensions: each is averaged on its own.
expect(judgePanelMean([panelScore(5, 1), panelScore(5, 1), panelScore(5, 1)], JUDGE_SCORE_DIMENSIONS).completeness).toBe(1);
});
test('malformed sample fields fail the panel instead of averaging to NaN', () => {
expect(() => judgePanelMean([panelScore(4), { ...panelScore(4), clarity: '4' as unknown as number }, panelScore(4)], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2 has non-numeric clarity');
expect(() => judgePanelMean([panelScore(4), null as unknown as ReturnType<typeof panelScore>], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2');
expect(() => judgePanelMean([], JUDGE_SCORE_DIMENSIONS)).toThrow('no samples');
});
test('boolean fields gate on a strict majority', () => {
const vote = (...values: boolean[]) => judgePanelMajority(values.map(value => ({ ok: value })), 'ok');
expect(vote(true, true, false)).toBe(true);
expect(vote(true, false, false)).toBe(false);
expect(vote(true, false)).toBe(false);
expect(() => judgePanelMajority([{ ok: true }, { ok: 'yes' }], 'ok')).toThrow('sample 2 has non-boolean ok');
});
test('reasoning keeps every sample, numbered, even for malformed samples', () => {
expect(judgePanelReasoning([panelScore(4), null, { reasoning: 7 }])).toBe('[sample 1] c4\n[sample 2] \n[sample 3] ');
});
test('the cache stores and validates only a complete panel against the mean', () => {
expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(true);
expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(2), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), { ...panelScore(4), clarity: 6 }] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4)], prompt: 'x' }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel(panelScore(4), panelThresholds)).toBe(false);
});
test('every judge in the quality file samples through the panel, never a lone call', () => {
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const calls = [...source.matchAll(/\b(?:callJudge<[^>(]*(?:<[^>]*>[^>(]*)*>|judge)\(/g)];
expect(calls.length).toBeGreaterThanOrEqual(8);
for (const call of calls) {
expect(source.slice(Math.max(0, call.index! - 25), call.index), `unpaneled judge call at offset ${call.index}`).toMatch(/judgePanel\(\(\) => $/);
}
expect(source).not.toMatch(/\bscores\.reasoning\b|\bresult\.reasoning\b/);
});
});