test(llm-judge): sample every judge as a pre-registered 3-sample panel

Each of the 24 skill-llm-eval judges now draws EVAL_POLICY.judge.samples
independent samples of the same prompt concurrently inside the unchanged
JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
against the unchanged threshold; booleans (would_browse, consistent) on a
strict majority. An erroring sample fails the whole panel and is never
resampled; a refusal is an unscored panel only when every sample refused.
callJudge's 429 backoff stays: it is transport before any model output.

The workflow-judge cache stores and validates only complete panels, and its
identity now records the panel and zero file retries. Harness tests that
pinned one provider call per case now pin the panel size.
This commit is contained in:
garrytan committed 2026-09-29 19:11:04 +00:00
1 parent adced7e046
commit 3f68572cab
7 files changed
+298 -112

No files matched your search

+146 -46
View File
@@ -1,18 +1,22 @@
import { afterEach, expect, spyOn, test } from 'bun:test';
import { afterEach, describe, expect, spyOn, test } from 'bun:test';
import { Messages } from '@anthropic-ai/sdk/resources/messages';
import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMajority, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge';
import { EVAL_POLICY } from './helpers/periodic-exclude-data';
import { getCookieWorkflowManualReview } from './helpers/cookie-workflow-manual-review';
import { resolveEvalModel } from '../lib/eval-model';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { execFileSync } from 'node:child_process';
import { prepareWorkflowJudgeCache, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
import { prepareWorkflowJudgeCache, validWorkflowJudgePanel, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA } from './helpers/workflow-judge-input';
const roots: string[] = [];
afterEach(() => { for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); });
const scores = { clarity: 4, completeness: 5, actionability: 4, reasoning: 'Concrete steps' };
const SAMPLES = JUDGE_PANEL_SAMPLES;
const panelOf = (sample: typeof scores) => Array.from({ length: SAMPLES }, () => sample);
const panel = panelOf(scores);
function fixture() {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-judge-cache-')); roots.push(root);
const files = {
@@ -53,9 +57,9 @@ function fixture() {
}
test('the audited adapter reuses only the exact completed score and original provenance', () => {
const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(scores);
const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(panel);
expect(f.entries()).toHaveLength(1);
const reused = f.cache().lookup(); expect(reused?.scores).toEqual(scores);
const reused = f.cache().lookup(); expect(reused?.samples).toEqual(panel);
expect(reused?.reuse.source.runId).toBe('free-cache-test');
expect(reused?.reuse.source.revision).toMatch(/^[a-f0-9]{40}$/);
expect(reused?.reuse.source.completedAt).toBeLessThanOrEqual(Date.now());
@@ -73,11 +77,11 @@ test('the dependency closure includes actual installed SDK bytes and local trans
});
test('release-label changes preserve reuse; other package semantics invalidate it', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
const file = path.join(f.root, 'package.json');
const original = JSON.parse(fs.readFileSync(file, 'utf8'));
fs.writeFileSync(file, JSON.stringify({ ...original, version: '2.0.0' }, null, 2));
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
for (const change of [{ scripts: { 'test:gate': 'changed command' } }, { dependencies: { 'some-sdk': '2.0.0' } }]) {
fs.writeFileSync(file, JSON.stringify({ ...original, ...change, version: '2.0.0' }));
expect(f.cache().lookup()).toBeNull();
@@ -89,7 +93,7 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in
'scripts/test-paid-shards.ts', 'scripts/test-strict-output.ts', 'scripts/eval-select.ts',
'scripts/test-pr-profile.ts', '.github/workflows/evals.yml']) {
test(`changes in ${file} require new evaluation`, () => {
const f = fixture(); f.cache().publish(scores); const target = path.join(f.root, file);
const f = fixture(); f.cache().publish(panel); const target = path.join(f.root, file);
fs.appendFileSync(target, file.endsWith('.json') ? ' ' : '\n// changed');
f.refreshPrompt(); expect(f.cache().lookup()).toBeNull();
});
@@ -98,8 +102,8 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in
test('changed sources during an attempt and mismatched actual prompt cannot publish', () => {
const f = fixture(); const before = f.cache();
fs.appendFileSync(path.join(f.root, 'example/sections/review.md'), 'new finding');
before.publish(scores); expect(f.entries()).toHaveLength(0);
f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(scores);
before.publish(panel); expect(f.entries()).toHaveLength(0);
f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(panel);
expect(f.entries()).toHaveLength(0);
});
@@ -108,52 +112,52 @@ for (const [key, value] of Object.entries({ EVALS_FRESH: '1', EVALS_TIER: 'perio
EVALS_CACHE_REPOSITORY: '', NODE_OPTIONS: '--require=unknown', BUN_OPTIONS: '--preload=unknown',
ANTHROPIC_BASE_URL: 'https://custom-provider.example.test' })) {
test(`${key}=${value} is fresh or ineligible`, () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.env = { ...f.env, [key]: value }; const cache = f.cache();
expect(cache.lookup()).toBeNull(); cache.publish(scores); expect(f.entries()).toHaveLength(1);
expect(cache.lookup()).toBeNull(); cache.publish(panel); expect(f.entries()).toHaveLength(1);
});
}
test('runtime/model/threshold changes miss, and retries never reuse or publish', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
for (const overrides of [{ GSTACK_EVAL_MODEL_JUDGE: 'different-model' }, { EVALS_CACHE_RUNTIME_ID: 'c'.repeat(64) }]) {
f.opts.env = { ...f.env, ...overrides }; expect(f.cache().lookup()).toBeNull();
}
f.opts.env = f.env; f.opts.thresholds.clarity = 5; expect(f.cache().lookup()).toBeNull();
f.opts.thresholds.clarity = 4; f.opts.attempt = 2; const retry = f.cache();
expect(retry.lookup()).toBeNull(); retry.publish(scores); expect(f.entries()).toHaveLength(1);
expect(retry.lookup()).toBeNull(); retry.publish(panel); expect(f.entries()).toHaveLength(1);
});
test('frontier reader calibration cannot reuse a score from the unspecified-reader rubric', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
const original = f.opts.prompt;
f.opts.agentCapability = 'frontier'; f.refreshPrompt();
expect(f.opts.prompt).not.toBe(original);
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(2);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
delete f.opts.agentCapability; f.refreshPrompt();
expect(f.opts.prompt).toBe(original);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('a pinned workflow judge model overrides the global model and changes the cache identity', () => {
const f = fixture();
f.opts.model = 'claude-sonnet-4-6';
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(1);
f.opts.env = { ...f.env, GSTACK_EVAL_MODEL_JUDGE: 'different-global-model' };
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
f.opts.model = 'claude-opus-4-7';
expect(f.cache().lookup()).toBeNull();
});
test('failed assertions, missing provenance, and missing imported dependencies cannot supply a receipt', () => {
const f = fixture(); f.cache().publish({ ...scores, clarity: 3 }); expect(f.entries()).toHaveLength(0);
f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(scores); expect(f.entries()).toHaveLength(0);
const f = fixture(); f.cache().publish(panelOf({ ...scores, clarity: 3 })); expect(f.entries()).toHaveLength(0);
f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(panel); expect(f.entries()).toHaveLength(0);
f.opts.env = f.env; fs.unlinkSync(path.join(f.root, 'test/helpers/nested.ts'));
f.cache().publish(scores); expect(f.entries()).toHaveLength(0);
f.cache().publish(panel); expect(f.entries()).toHaveLength(0);
});
test('cached payload schema remains small and cannot carry operational fields', () => {
@@ -167,8 +171,9 @@ test('workflow registration preserves model work and reserves only terminal-reco
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const body = source.split('async function runWorkflowJudge')[1]!.split('// Block 1:')[0]!;
const stages = ['workflowJudgeAttempts.set', 'readWorkflowJudgeInput(', 'cache.lookup()',
'callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(scores, active)']
'judgePanel(() => callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
'scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);',
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(samples, active)']
.map(stage => body.indexOf(stage));
expect(stages.every(position => position >= 0)).toBe(true);
expect(stages).toEqual([...stages].sort((a, b) => a - b));
@@ -203,6 +208,7 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
'evalCollector', 'expect', 'console', 'performance', 'JUDGE_MS', 'WORKFLOW_JUDGE_RECORD_MS',
'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'resolveEvalModel',
'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore',
'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS',
`${javascript}\nreturn runWorkflowJudge;`)(
f.root, overrides.read ?? readWorkflowJudgeInput, buildWorkflowJudgePrompt,
(options: WorkflowCacheOptions) => (overrides.prepare ?? prepareWorkflowJudgeCache)({ ...options, env: f.env }),
@@ -213,7 +219,8 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
overrides.clock ? { now: overrides.clock } : performance, overrides.budget ?? 120_000, overrides.allowance ?? 5_000,
overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout,
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore);
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore,
judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS);
return { run, records, signals, prompts, attempts, options: { ...f.opts, suite: 'Cache regression' } };
}
@@ -223,7 +230,7 @@ test('the actual workflow callback preserves the pinned model and frontier rubri
const actual = actualCallback(f, { judge: async (_prompt, model) => { models.push(model); return scores; } });
await actual.run({ ...actual.options, model: 'claude-sonnet-4-6', agentCapability: 'frontier',
readInput: () => readWorkflowJudgeInput(f.opts) });
expect(models).toEqual(['claude-sonnet-4-6']);
expect(models).toEqual(Array(SAMPLES).fill('claude-sonnet-4-6'));
expect(actual.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
expect(actual.records[0]).toMatchObject({ passed: true, model: 'claude-sonnet-4-6', prompt: actual.prompts[0] });
});
@@ -239,7 +246,7 @@ test.each(['ship', 'review'])('the registered %s callback sends the frontier rub
endMarker: f.opts.endMarker, references: [] };
const passing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 3 }) });
await passing.run(options);
expect(passing.prompts).toHaveLength(1);
expect(passing.prompts).toHaveLength(SAMPLES);
expect(passing.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
expect(passing.records[0]).toMatchObject({ passed: true, execution: 'executed', judge_scores: { clarity: 3 } });
const failing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 2 }) });
@@ -254,8 +261,8 @@ test('the actual workflow callback executes once, reuses with provenance, and pr
const f = fixture(); const first = actualCallback(f);
const options = { ...f.opts, suite: 'Cache regression' };
await first.run(options);
expect(first.prompts).toEqual([f.opts.prompt]); expect(f.entries()).toHaveLength(1);
expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 });
expect(first.prompts).toEqual(Array(SAMPLES).fill(f.opts.prompt)); expect(f.entries()).toHaveLength(1);
expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 * SAMPLES });
expect(first.records[0]).not.toHaveProperty('prompt');
expect(first.records[0]).not.toHaveProperty('model');
const reused = actualCallback(f, { judge: async () => ({ ...scores, clarity: 1 }) });
@@ -312,13 +319,13 @@ test('a superseding attempt cancels its predecessor before either can record a s
});
test('a failed input read consumes attempt one and prevents a retry from borrowing or publishing a receipt', async () => {
const f = fixture(); f.cache().publish(scores); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8');
const f = fixture(); f.cache().publish(panel); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8');
let reads = 0;
const h = actualCallback(f, { read: options => { if (++reads === 1) throw new Error('Missing workflow fixture'); return readWorkflowJudgeInput(options); } });
await expect(h.run(h.options)).rejects.toThrow('Missing workflow fixture');
expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'harness_error' });
await h.run(h.options);
expect(h.prompts).toHaveLength(1);
expect(h.prompts).toHaveLength(SAMPLES);
expect(h.records.map(record => record.execution)).toEqual(['executed', 'executed']);
expect(h.attempts.get(f.opts.testName).attempt).toBe(2);
expect(fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8')).toBe(receipt);
@@ -333,14 +340,14 @@ test('monotonic expiry after a synchronous preparation or late model response re
await expect(h.run(h.options)).rejects.toThrow('deadline');
expect(h.records).toHaveLength(1);
expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'timeout', duration_ms: 21 });
expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : 1);
expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : SAMPLES);
expect(f.entries()).toHaveLength(0);
}
});
test('publication rechecks after input scanning and withdraws a receipt if recording expires', async () => {
const f = fixture(); let checks = 0;
f.cache().publish(scores, () => ++checks < 2);
f.cache().publish(panel, () => ++checks < 2);
expect(checks).toBe(2); expect(f.entries()).toHaveLength(0);
let now = 0;
const h = actualCallback(f, { budget: 20, allowance: 5, clock: () => now,
@@ -373,7 +380,7 @@ test('the actual workflow callback preserves the complete public API body; cance
try {
const h = actualCallback(f, { judge: (prompt, model, options) => callJudge<typeof scores>(prompt, model, options) });
await h.run(h.options);
expect(create).toHaveBeenCalledTimes(1);
expect(create).toHaveBeenCalledTimes(SAMPLES);
expect(create.mock.calls[0]).toEqual([{
model: resolveEvalModel('judge'), max_tokens: 8192,
messages: [{ role: 'user', content: f.opts.prompt }],
@@ -412,40 +419,40 @@ test('Ship sends its authorized 64k cap and compact response contract through th
f.opts.structuredResponse = true;
f.opts.maxTokens = 65_536;
f.opts.stream = true;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
} finally { stream.mockRestore(); }
});
test('changing response serialization misses the cache even when prompt and model match', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.structuredResponse = true;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(2);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
const description = WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description;
try {
WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description += ' Changed response contract.';
expect(f.cache().lookup()).toBeNull();
} finally { WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description = description; }
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
f.opts.structuredResponse = false;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('the actual cap and streaming transport independently affect workflow cache identity', () => {
const f = fixture(); f.cache().publish(scores);
const f = fixture(); f.cache().publish(panel);
f.opts.maxTokens = 65_536;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
f.opts.stream = true;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.cache().publish(panel);
expect(f.entries()).toHaveLength(3);
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
delete f.opts.maxTokens;
delete f.opts.stream;
expect(f.cache().lookup()?.scores).toEqual(scores);
expect(f.cache().lookup()?.samples).toEqual(panel);
});
test('the structured callback rejects incomplete, schema-invalid and below-threshold answers without cache credit', async () => {
@@ -473,3 +480,96 @@ test('the structured callback rejects incomplete, schema-invalid and below-thres
expect(validWorkflowJudgeScore({ ...scores, reasoning: Array(149).fill('word').join(' ') }, { clarity: 1, completeness: 1, actionability: 1 }, true)).toBe(true);
} finally { stream.mockRestore(); diagnostics.mockRestore(); }
});
// --- Judge panel policy (EVAL_POLICY.judge): fixed concurrent samples, per-dimension
// mean and boolean majority against unchanged thresholds, an erroring sample fails
// the whole panel and is never resampled. The provider is always a stub.
const panelScore = (clarity: number, completeness = 4, actionability = 4) => ({ clarity, completeness, actionability, reasoning: `c${clarity}` });
const panelThresholds = { clarity: 3, completeness: 3, actionability: 4 };
const panelRefusal = () => new JudgeRefusalError({ id: 'msg_1', _request_id: 'req_1', model: 'm', usage: { input_tokens: 1, output_tokens: 0 }, content: [] });
describe('judge panel', () => {
test('the pre-registered panel is three samples, and the helper restates EVAL_POLICY exactly', () => {
expect(EVAL_POLICY.judge.samples).toBe(3);
expect(JUDGE_PANEL_SAMPLES).toBe(EVAL_POLICY.judge.samples);
});
test('draws every sample concurrently before any resolves', async () => {
let started = 0;
const releases: Array<() => void> = [];
const panel = judgePanel(() => new Promise<number>(resolve => { started += 1; releases.push(() => resolve(started)); }));
await Promise.resolve();
expect(started).toBe(SAMPLES);
releases.forEach(release => release());
expect(await panel).toHaveLength(SAMPLES);
});
test('an erroring sample fails the panel and is never resampled', async () => {
let calls = 0;
const panel = judgePanel(async () => {
calls += 1;
if (calls === 2) throw new Error('Judge returned non-JSON: nope');
return panelScore(5);
});
await expect(panel).rejects.toThrow('non-JSON');
expect(calls).toBe(SAMPLES);
});
test('a refusal on every sample stays a provider refusal; a partial refusal is an ordinary failure', async () => {
await expect(judgePanel(async () => { throw panelRefusal(); })).rejects.toBeInstanceOf(JudgeRefusalError);
let calls = 0;
const partial = judgePanel(async () => { if (++calls === 1) throw panelRefusal(); return panelScore(4); });
const error = await partial.then(() => null, (reason: unknown) => reason);
expect(error).toBeInstanceOf(Error);
expect(error).not.toBeInstanceOf(JudgeRefusalError);
expect(String(error)).toContain(`sample 1 of ${SAMPLES} failed beside scored samples`);
});
test('numeric dimensions gate on the per-dimension mean; one low sample can be outvoted, a low mean cannot', () => {
const outvoted = judgePanelMean([panelScore(2), panelScore(4), panelScore(4)], JUDGE_SCORE_DIMENSIONS);
expect(outvoted.clarity).toBeCloseTo(10 / 3);
expect(outvoted.clarity).toBeGreaterThanOrEqual(panelThresholds.clarity);
const low = judgePanelMean([panelScore(2), panelScore(2), panelScore(4)], JUDGE_SCORE_DIMENSIONS);
expect(low.clarity).toBeLessThan(panelThresholds.clarity);
// No compensation across dimensions: each is averaged on its own.
expect(judgePanelMean([panelScore(5, 1), panelScore(5, 1), panelScore(5, 1)], JUDGE_SCORE_DIMENSIONS).completeness).toBe(1);
});
test('malformed sample fields fail the panel instead of averaging to NaN', () => {
expect(() => judgePanelMean([panelScore(4), { ...panelScore(4), clarity: '4' as unknown as number }, panelScore(4)], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2 has non-numeric clarity');
expect(() => judgePanelMean([panelScore(4), null as unknown as ReturnType<typeof panelScore>], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2');
expect(() => judgePanelMean([], JUDGE_SCORE_DIMENSIONS)).toThrow('no samples');
});
test('boolean fields gate on a strict majority', () => {
const vote = (...values: boolean[]) => judgePanelMajority(values.map(value => ({ ok: value })), 'ok');
expect(vote(true, true, false)).toBe(true);
expect(vote(true, false, false)).toBe(false);
expect(vote(true, false)).toBe(false);
expect(() => judgePanelMajority([{ ok: true }, { ok: 'yes' }], 'ok')).toThrow('sample 2 has non-boolean ok');
});
test('reasoning keeps every sample, numbered, even for malformed samples', () => {
expect(judgePanelReasoning([panelScore(4), null, { reasoning: 7 }])).toBe('[sample 1] c4\n[sample 2] \n[sample 3] ');
});
test('the cache stores and validates only a complete panel against the mean', () => {
expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(true);
expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(2), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), { ...panelScore(4), clarity: 6 }] }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4)], prompt: 'x' }, panelThresholds)).toBe(false);
expect(validWorkflowJudgePanel(panelScore(4), panelThresholds)).toBe(false);
});
test('every judge in the quality file samples through the panel, never a lone call', () => {
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const calls = [...source.matchAll(/\b(?:callJudge<[^>(]*(?:<[^>]*>[^>(]*)*>|judge)\(/g)];
expect(calls.length).toBeGreaterThanOrEqual(8);
for (const call of calls) {
expect(source.slice(Math.max(0, call.index! - 25), call.index), `unpaneled judge call at offset ${call.index}`).toMatch(/judgePanel\(\(\) => $/);
}
expect(source).not.toMatch(/\bscores\.reasoning\b|\bresult\.reasoning\b/);
});
});