diff --git a/test/arm-benchmark-selftest.test.ts b/test/arm-benchmark-selftest.test.ts index e622bf645..347d1e8f5 100644 --- a/test/arm-benchmark-selftest.test.ts +++ b/test/arm-benchmark-selftest.test.ts @@ -13,7 +13,7 @@ import { } from './helpers/arm-benchmark-harness'; import { armJudge, buildArmJudgePrompt, parseArmJudgeResponse, - ARM_JUDGE_ATTEMPTS, callJudge, + callJudge, } from './helpers/llm-judge'; import * as fs from 'fs'; import * as path from 'path'; @@ -182,28 +182,25 @@ describe('arm benchmark selftest (free, no API)', () => { expect(score.construct).toBe('none'); }); - test('armJudge: bounded retry-on-malformed — recovers once, then gives up', async () => { - // Malformed first, valid second: recovers within the 2-attempt bound. + test('armJudge: a malformed verdict is a failed sample, never re-asked', async () => { let calls = 0; - const flaky = (async () => { + const malformedFirst = (async () => { calls++; return calls === 1 ? { over_engineering: 9, construct: 'garbage' } : { over_engineering: 2, construct: 'repository layer in app.js', reasoning: 'ok' }; }) as unknown as typeof callJudge; - const recovered = await armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: flaky }); - expect(recovered.over_engineering).toBe(2); - expect(calls).toBe(ARM_JUDGE_ATTEMPTS); + await expect(armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: malformedFirst })) + .rejects.toThrow(/malformed verdict \(never resampled\)/); + expect(calls).toBe(1); - // Always malformed: throws after exactly ARM_JUDGE_ATTEMPTS attempts. - let badCalls = 0; - const alwaysBad = (async () => { - badCalls++; - return { nonsense: true }; + let goodCalls = 0; + const wellFormed = (async () => { + goodCalls++; + return { over_engineering: 2, construct: 'repository layer in app.js', reasoning: 'ok' }; }) as unknown as typeof callJudge; - await expect(armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: alwaysBad })) - .rejects.toThrow(/no well-formed verdict after 2 attempts/); - expect(badCalls).toBe(ARM_JUDGE_ATTEMPTS); + expect((await armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: wellFormed })).over_engineering).toBe(2); + expect(goodCalls).toBe(1); }); }); diff --git a/test/helpers/llm-judge.ts b/test/helpers/llm-judge.ts index da7811b47..22c24a973 100644 --- a/test/helpers/llm-judge.ts +++ b/test/helpers/llm-judge.ts @@ -512,9 +512,6 @@ export interface ArmJudgeScore { */ export const ARM_JUDGE_MODEL = CLAUDE_FRONTIER_EVAL_MODEL; -/** Bounded retry-on-malformed loop: total attempts, not extra retries. */ -export const ARM_JUDGE_ATTEMPTS = 2; - /** * Build the over-engineering rubric prompt. Exported (pure) so the free * selftest can verify prompt construction without any API call. @@ -587,10 +584,10 @@ export function parseArmJudgeResponse(raw: unknown): ArmJudgeScore { * * - Zero-diff arms are VALID scored cells: the agent built nothing, so the * score is deterministically 0/"none" — no API call. - * - Bounded retry-on-malformed: ARM_JUDGE_ATTEMPTS total attempts. callJudge - * already retries 429s internally; this loop covers malformed/refused JSON. + * - One sample, never re-asked: a malformed or refused verdict is a failed + * sample. callJudge's transport-level 429 backoff is not a verdict retry. * - `opts.call` is an injection seam so the free selftest can exercise the - * retry bound without spending API money. Defaults to the real callJudge. + * malformed path without spending API money. Defaults to the real callJudge. */ export async function armJudge( task: string, @@ -605,18 +602,10 @@ export async function armJudge( }; } const call = opts?.call ?? callJudge; - const prompt = buildArmJudgePrompt(task, diff); - let lastError: unknown; - for (let attempt = 1; attempt <= ARM_JUDGE_ATTEMPTS; attempt++) { - try { - const raw = await call>(prompt, ARM_JUDGE_MODEL); - return parseArmJudgeResponse(raw); - } catch (err) { - lastError = err; - } + const raw = await call>(buildArmJudgePrompt(task, diff), ARM_JUDGE_MODEL); + try { + return parseArmJudgeResponse(raw); + } catch (err) { + throw new Error(`armJudge: malformed verdict (never resampled) — ${err instanceof Error ? err.message : String(err)}`); } - throw new Error( - `armJudge: no well-formed verdict after ${ARM_JUDGE_ATTEMPTS} attempts — ` - + (lastError instanceof Error ? lastError.message : String(lastError)), - ); } diff --git a/test/llm-judge-recommendation.test.ts b/test/llm-judge-recommendation.test.ts index 438d1da37..05af47a05 100644 --- a/test/llm-judge-recommendation.test.ts +++ b/test/llm-judge-recommendation.test.ts @@ -6,14 +6,16 @@ * negative coverage: hand-graded good/bad recommendation strings, asserted * against the same threshold the production E2E tests use (>= 4). * - * Costs ~$0.04 per run (4 Haiku calls + 3 deterministic-only fixtures). + * Each fixture is a pre-registered 3-sample judge panel: numeric substance + * gates on the panel mean, the boolean checks on a 2-of-3 majority, and an + * erroring sample fails the panel (never resampled). Costs ~$0.12 per run. * Touchfile-gated to test/helpers/llm-judge.ts so it fires on rubric * tweaks but not every test run. Runs only under EVALS=1 with an API key. */ import { expect } from 'bun:test'; import { CAPTURE_MS } from './helpers/eval-budgets'; -import { judgeRecommendation } from './helpers/llm-judge'; +import { judgePanel, judgePanelMajority, judgePanelMean, judgePanelReasoning, judgeRecommendation } from './helpers/llm-judge'; import { describeIfSelected, testIfSelected } from './helpers/e2e-helpers'; // Fixtures wrap a realistic AskUserQuestion shape so the judge sees the menu @@ -37,13 +39,24 @@ C) Hybrid — V1 client-side, V1.5 promotes to gbrain Net: optimize for V1 ship velocity vs long-term agent reusability.`; } +async function judgeRecommendationPanel(text: string) { + const samples = await judgePanel(() => judgeRecommendation(text)); + return { + present: judgePanelMajority(samples, 'present'), + commits: judgePanelMajority(samples, 'commits'), + has_because: judgePanelMajority(samples, 'has_because'), + reason_substance: judgePanelMean(samples, ['reason_substance']).reason_substance, + reasoning: judgePanelReasoning(samples), + }; +} + describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendation'], () => { testIfSelected('llm-judge-recommendation', async () => { // Run all 7 fixtures sequentially in one test entry so the eval-store sees // a single result; individual assertions surface as failed expectations. // SUBSTANCE 5: option-specific reason that contrasts an alternative. - const good5 = await judgeRecommendation(buildAUQ( + const good5 = await judgeRecommendationPanel(buildAUQ( 'Recommendation: Choose C because hybrid ships V1 in gstack-only without blocking on cross-repo gbrain coordination, and locks the migration path before other agents take a hard dependency.', )); expect(good5.present).toBe(true); @@ -55,7 +68,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati ).toBeGreaterThanOrEqual(4); // SUBSTANCE 4: concrete option-specific reason without alternative comparison. - const good4 = await judgeRecommendation(buildAUQ( + const good4 = await judgeRecommendationPanel(buildAUQ( 'Recommendation: Choose B because client-side composition uses MCP tools that already exist in gstack and avoids any gbrain release dependency for V1.', )); expect(good4.present).toBe(true); @@ -65,7 +78,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati ).toBeGreaterThanOrEqual(4); // SUBSTANCE ~1: boilerplate. - const bad1 = await judgeRecommendation(buildAUQ( + const bad1 = await judgeRecommendationPanel(buildAUQ( 'Recommendation: Choose B because it is better.', )); expect(bad1.present).toBe(true); @@ -76,7 +89,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati ).toBeLessThan(4); // SUBSTANCE ~3: generic. - const bad3 = await judgeRecommendation(buildAUQ( + const bad3 = await judgeRecommendationPanel(buildAUQ( 'Recommendation: Choose B because it is faster.', )); expect(bad3.present).toBe(true); @@ -87,7 +100,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati ).toBeLessThan(4); // NO BECAUSE: missing causal connective. - const noBecause = await judgeRecommendation(buildAUQ( + const noBecause = await judgeRecommendationPanel(buildAUQ( 'Recommendation: Choose B (it has the best tradeoffs).', )); expect(noBecause.present).toBe(true); @@ -95,7 +108,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati expect(noBecause.reason_substance).toBe(1); // NO RECOMMENDATION: line missing entirely. - const noRec = await judgeRecommendation(`D1 — Where should the smarts live? + const noRec = await judgeRecommendationPanel(`D1 — Where should the smarts live? ELI10: ... Pros / cons: A) Server-side @@ -146,7 +159,7 @@ Net: ...`); ], ] as Array<[string, string, boolean]>; for (const [label, text, shouldPass] of crossModelCases) { - const score = await judgeRecommendation(text); + const score = await judgeRecommendationPanel(text); expect(score.present, `[cross-model:${label}] present should be true`).toBe(true); expect(score.has_because, `[cross-model:${label}] has_because should be true`).toBe(true); if (shouldPass) { @@ -175,7 +188,7 @@ Net: ...`); ['whichever fits', 'Recommendation: whichever fits the team — A or B both work.'], ]; for (const [label, text] of hedgeForms) { - const score = await judgeRecommendation(buildAUQ(text)); + const score = await judgeRecommendationPanel(buildAUQ(text)); expect(score.present, `[hedge:${label}] present should be true`).toBe(true); expect( score.commits,