mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
test(judges): sample the recommendation rubric as a panel; never re-ask armJudge
llm-judge-recommendation is a judge case: each fixture now draws a 3-sample judgePanel, gates reason_substance on the panel mean and the present/commits/has_because checks on a 2-of-3 majority, thresholds unchanged. armJudge no longer re-asks on a malformed verdict; it is a failed sample, as the judge policy requires.
This commit is contained in:
1 parent
2fb55e4c7a
commit
f4ab5ee75f
3 files changed
+43
-44
No files matched your search
@@ -13,7 +13,7 @@ import {
|
|||||||
} from './helpers/arm-benchmark-harness';
|
} from './helpers/arm-benchmark-harness';
|
||||||
import {
|
import {
|
||||||
armJudge, buildArmJudgePrompt, parseArmJudgeResponse,
|
armJudge, buildArmJudgePrompt, parseArmJudgeResponse,
|
||||||
ARM_JUDGE_ATTEMPTS, callJudge,
|
callJudge,
|
||||||
} from './helpers/llm-judge';
|
} from './helpers/llm-judge';
|
||||||
import * as fs from 'fs';
|
import * as fs from 'fs';
|
||||||
import * as path from 'path';
|
import * as path from 'path';
|
||||||
@@ -182,28 +182,25 @@ describe('arm benchmark selftest (free, no API)', () => {
|
|||||||
expect(score.construct).toBe('none');
|
expect(score.construct).toBe('none');
|
||||||
});
|
});
|
||||||
|
|
||||||
test('armJudge: bounded retry-on-malformed — recovers once, then gives up', async () => {
|
test('armJudge: a malformed verdict is a failed sample, never re-asked', async () => {
|
||||||
// Malformed first, valid second: recovers within the 2-attempt bound.
|
|
||||||
let calls = 0;
|
let calls = 0;
|
||||||
const flaky = (async () => {
|
const malformedFirst = (async () => {
|
||||||
calls++;
|
calls++;
|
||||||
return calls === 1
|
return calls === 1
|
||||||
? { over_engineering: 9, construct: 'garbage' }
|
? { over_engineering: 9, construct: 'garbage' }
|
||||||
: { over_engineering: 2, construct: 'repository layer in app.js', reasoning: 'ok' };
|
: { over_engineering: 2, construct: 'repository layer in app.js', reasoning: 'ok' };
|
||||||
}) as unknown as typeof callJudge;
|
}) as unknown as typeof callJudge;
|
||||||
const recovered = await armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: flaky });
|
await expect(armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: malformedFirst }))
|
||||||
expect(recovered.over_engineering).toBe(2);
|
.rejects.toThrow(/malformed verdict \(never resampled\)/);
|
||||||
expect(calls).toBe(ARM_JUDGE_ATTEMPTS);
|
expect(calls).toBe(1);
|
||||||
|
|
||||||
// Always malformed: throws after exactly ARM_JUDGE_ATTEMPTS attempts.
|
let goodCalls = 0;
|
||||||
let badCalls = 0;
|
const wellFormed = (async () => {
|
||||||
const alwaysBad = (async () => {
|
goodCalls++;
|
||||||
badCalls++;
|
return { over_engineering: 2, construct: 'repository layer in app.js', reasoning: 'ok' };
|
||||||
return { nonsense: true };
|
|
||||||
}) as unknown as typeof callJudge;
|
}) as unknown as typeof callJudge;
|
||||||
await expect(armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: alwaysBad }))
|
expect((await armJudge('ticket', 'diff --git a/x b/x\n+1\n', { call: wellFormed })).over_engineering).toBe(2);
|
||||||
.rejects.toThrow(/no well-formed verdict after 2 attempts/);
|
expect(goodCalls).toBe(1);
|
||||||
expect(badCalls).toBe(ARM_JUDGE_ATTEMPTS);
|
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|||||||
@@ -512,9 +512,6 @@ export interface ArmJudgeScore {
|
|||||||
*/
|
*/
|
||||||
export const ARM_JUDGE_MODEL = CLAUDE_FRONTIER_EVAL_MODEL;
|
export const ARM_JUDGE_MODEL = CLAUDE_FRONTIER_EVAL_MODEL;
|
||||||
|
|
||||||
/** Bounded retry-on-malformed loop: total attempts, not extra retries. */
|
|
||||||
export const ARM_JUDGE_ATTEMPTS = 2;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Build the over-engineering rubric prompt. Exported (pure) so the free
|
* Build the over-engineering rubric prompt. Exported (pure) so the free
|
||||||
* selftest can verify prompt construction without any API call.
|
* selftest can verify prompt construction without any API call.
|
||||||
@@ -587,10 +584,10 @@ export function parseArmJudgeResponse(raw: unknown): ArmJudgeScore {
|
|||||||
*
|
*
|
||||||
* - Zero-diff arms are VALID scored cells: the agent built nothing, so the
|
* - Zero-diff arms are VALID scored cells: the agent built nothing, so the
|
||||||
* score is deterministically 0/"none" — no API call.
|
* score is deterministically 0/"none" — no API call.
|
||||||
* - Bounded retry-on-malformed: ARM_JUDGE_ATTEMPTS total attempts. callJudge
|
* - One sample, never re-asked: a malformed or refused verdict is a failed
|
||||||
* already retries 429s internally; this loop covers malformed/refused JSON.
|
* sample. callJudge's transport-level 429 backoff is not a verdict retry.
|
||||||
* - `opts.call` is an injection seam so the free selftest can exercise the
|
* - `opts.call` is an injection seam so the free selftest can exercise the
|
||||||
* retry bound without spending API money. Defaults to the real callJudge.
|
* malformed path without spending API money. Defaults to the real callJudge.
|
||||||
*/
|
*/
|
||||||
export async function armJudge(
|
export async function armJudge(
|
||||||
task: string,
|
task: string,
|
||||||
@@ -605,18 +602,10 @@ export async function armJudge(
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
const call = opts?.call ?? callJudge;
|
const call = opts?.call ?? callJudge;
|
||||||
const prompt = buildArmJudgePrompt(task, diff);
|
const raw = await call<Record<string, unknown>>(buildArmJudgePrompt(task, diff), ARM_JUDGE_MODEL);
|
||||||
let lastError: unknown;
|
try {
|
||||||
for (let attempt = 1; attempt <= ARM_JUDGE_ATTEMPTS; attempt++) {
|
return parseArmJudgeResponse(raw);
|
||||||
try {
|
} catch (err) {
|
||||||
const raw = await call<Record<string, unknown>>(prompt, ARM_JUDGE_MODEL);
|
throw new Error(`armJudge: malformed verdict (never resampled) — ${err instanceof Error ? err.message : String(err)}`);
|
||||||
return parseArmJudgeResponse(raw);
|
|
||||||
} catch (err) {
|
|
||||||
lastError = err;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
throw new Error(
|
|
||||||
`armJudge: no well-formed verdict after ${ARM_JUDGE_ATTEMPTS} attempts — `
|
|
||||||
+ (lastError instanceof Error ? lastError.message : String(lastError)),
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
@@ -6,14 +6,16 @@
|
|||||||
* negative coverage: hand-graded good/bad recommendation strings, asserted
|
* negative coverage: hand-graded good/bad recommendation strings, asserted
|
||||||
* against the same threshold the production E2E tests use (>= 4).
|
* against the same threshold the production E2E tests use (>= 4).
|
||||||
*
|
*
|
||||||
* Costs ~$0.04 per run (4 Haiku calls + 3 deterministic-only fixtures).
|
* Each fixture is a pre-registered 3-sample judge panel: numeric substance
|
||||||
|
* gates on the panel mean, the boolean checks on a 2-of-3 majority, and an
|
||||||
|
* erroring sample fails the panel (never resampled). Costs ~$0.12 per run.
|
||||||
* Touchfile-gated to test/helpers/llm-judge.ts so it fires on rubric
|
* Touchfile-gated to test/helpers/llm-judge.ts so it fires on rubric
|
||||||
* tweaks but not every test run. Runs only under EVALS=1 with an API key.
|
* tweaks but not every test run. Runs only under EVALS=1 with an API key.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
import { expect } from 'bun:test';
|
import { expect } from 'bun:test';
|
||||||
import { CAPTURE_MS } from './helpers/eval-budgets';
|
import { CAPTURE_MS } from './helpers/eval-budgets';
|
||||||
import { judgeRecommendation } from './helpers/llm-judge';
|
import { judgePanel, judgePanelMajority, judgePanelMean, judgePanelReasoning, judgeRecommendation } from './helpers/llm-judge';
|
||||||
import { describeIfSelected, testIfSelected } from './helpers/e2e-helpers';
|
import { describeIfSelected, testIfSelected } from './helpers/e2e-helpers';
|
||||||
|
|
||||||
// Fixtures wrap a realistic AskUserQuestion shape so the judge sees the menu
|
// Fixtures wrap a realistic AskUserQuestion shape so the judge sees the menu
|
||||||
@@ -37,13 +39,24 @@ C) Hybrid — V1 client-side, V1.5 promotes to gbrain
|
|||||||
Net: optimize for V1 ship velocity vs long-term agent reusability.`;
|
Net: optimize for V1 ship velocity vs long-term agent reusability.`;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async function judgeRecommendationPanel(text: string) {
|
||||||
|
const samples = await judgePanel(() => judgeRecommendation(text));
|
||||||
|
return {
|
||||||
|
present: judgePanelMajority(samples, 'present'),
|
||||||
|
commits: judgePanelMajority(samples, 'commits'),
|
||||||
|
has_because: judgePanelMajority(samples, 'has_because'),
|
||||||
|
reason_substance: judgePanelMean(samples, ['reason_substance']).reason_substance,
|
||||||
|
reasoning: judgePanelReasoning(samples),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendation'], () => {
|
describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendation'], () => {
|
||||||
testIfSelected('llm-judge-recommendation', async () => {
|
testIfSelected('llm-judge-recommendation', async () => {
|
||||||
// Run all 7 fixtures sequentially in one test entry so the eval-store sees
|
// Run all 7 fixtures sequentially in one test entry so the eval-store sees
|
||||||
// a single result; individual assertions surface as failed expectations.
|
// a single result; individual assertions surface as failed expectations.
|
||||||
|
|
||||||
// SUBSTANCE 5: option-specific reason that contrasts an alternative.
|
// SUBSTANCE 5: option-specific reason that contrasts an alternative.
|
||||||
const good5 = await judgeRecommendation(buildAUQ(
|
const good5 = await judgeRecommendationPanel(buildAUQ(
|
||||||
'Recommendation: Choose C because hybrid ships V1 in gstack-only without blocking on cross-repo gbrain coordination, and locks the migration path before other agents take a hard dependency.',
|
'Recommendation: Choose C because hybrid ships V1 in gstack-only without blocking on cross-repo gbrain coordination, and locks the migration path before other agents take a hard dependency.',
|
||||||
));
|
));
|
||||||
expect(good5.present).toBe(true);
|
expect(good5.present).toBe(true);
|
||||||
@@ -55,7 +68,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati
|
|||||||
).toBeGreaterThanOrEqual(4);
|
).toBeGreaterThanOrEqual(4);
|
||||||
|
|
||||||
// SUBSTANCE 4: concrete option-specific reason without alternative comparison.
|
// SUBSTANCE 4: concrete option-specific reason without alternative comparison.
|
||||||
const good4 = await judgeRecommendation(buildAUQ(
|
const good4 = await judgeRecommendationPanel(buildAUQ(
|
||||||
'Recommendation: Choose B because client-side composition uses MCP tools that already exist in gstack and avoids any gbrain release dependency for V1.',
|
'Recommendation: Choose B because client-side composition uses MCP tools that already exist in gstack and avoids any gbrain release dependency for V1.',
|
||||||
));
|
));
|
||||||
expect(good4.present).toBe(true);
|
expect(good4.present).toBe(true);
|
||||||
@@ -65,7 +78,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati
|
|||||||
).toBeGreaterThanOrEqual(4);
|
).toBeGreaterThanOrEqual(4);
|
||||||
|
|
||||||
// SUBSTANCE ~1: boilerplate.
|
// SUBSTANCE ~1: boilerplate.
|
||||||
const bad1 = await judgeRecommendation(buildAUQ(
|
const bad1 = await judgeRecommendationPanel(buildAUQ(
|
||||||
'Recommendation: Choose B because it is better.',
|
'Recommendation: Choose B because it is better.',
|
||||||
));
|
));
|
||||||
expect(bad1.present).toBe(true);
|
expect(bad1.present).toBe(true);
|
||||||
@@ -76,7 +89,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati
|
|||||||
).toBeLessThan(4);
|
).toBeLessThan(4);
|
||||||
|
|
||||||
// SUBSTANCE ~3: generic.
|
// SUBSTANCE ~3: generic.
|
||||||
const bad3 = await judgeRecommendation(buildAUQ(
|
const bad3 = await judgeRecommendationPanel(buildAUQ(
|
||||||
'Recommendation: Choose B because it is faster.',
|
'Recommendation: Choose B because it is faster.',
|
||||||
));
|
));
|
||||||
expect(bad3.present).toBe(true);
|
expect(bad3.present).toBe(true);
|
||||||
@@ -87,7 +100,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati
|
|||||||
).toBeLessThan(4);
|
).toBeLessThan(4);
|
||||||
|
|
||||||
// NO BECAUSE: missing causal connective.
|
// NO BECAUSE: missing causal connective.
|
||||||
const noBecause = await judgeRecommendation(buildAUQ(
|
const noBecause = await judgeRecommendationPanel(buildAUQ(
|
||||||
'Recommendation: Choose B (it has the best tradeoffs).',
|
'Recommendation: Choose B (it has the best tradeoffs).',
|
||||||
));
|
));
|
||||||
expect(noBecause.present).toBe(true);
|
expect(noBecause.present).toBe(true);
|
||||||
@@ -95,7 +108,7 @@ describeIfSelected('judgeRecommendation rubric sanity', ['llm-judge-recommendati
|
|||||||
expect(noBecause.reason_substance).toBe(1);
|
expect(noBecause.reason_substance).toBe(1);
|
||||||
|
|
||||||
// NO RECOMMENDATION: line missing entirely.
|
// NO RECOMMENDATION: line missing entirely.
|
||||||
const noRec = await judgeRecommendation(`D1 — Where should the smarts live?
|
const noRec = await judgeRecommendationPanel(`D1 — Where should the smarts live?
|
||||||
ELI10: ...
|
ELI10: ...
|
||||||
Pros / cons:
|
Pros / cons:
|
||||||
A) Server-side
|
A) Server-side
|
||||||
@@ -146,7 +159,7 @@ Net: ...`);
|
|||||||
],
|
],
|
||||||
] as Array<[string, string, boolean]>;
|
] as Array<[string, string, boolean]>;
|
||||||
for (const [label, text, shouldPass] of crossModelCases) {
|
for (const [label, text, shouldPass] of crossModelCases) {
|
||||||
const score = await judgeRecommendation(text);
|
const score = await judgeRecommendationPanel(text);
|
||||||
expect(score.present, `[cross-model:${label}] present should be true`).toBe(true);
|
expect(score.present, `[cross-model:${label}] present should be true`).toBe(true);
|
||||||
expect(score.has_because, `[cross-model:${label}] has_because should be true`).toBe(true);
|
expect(score.has_because, `[cross-model:${label}] has_because should be true`).toBe(true);
|
||||||
if (shouldPass) {
|
if (shouldPass) {
|
||||||
@@ -175,7 +188,7 @@ Net: ...`);
|
|||||||
['whichever fits', 'Recommendation: whichever fits the team — A or B both work.'],
|
['whichever fits', 'Recommendation: whichever fits the team — A or B both work.'],
|
||||||
];
|
];
|
||||||
for (const [label, text] of hedgeForms) {
|
for (const [label, text] of hedgeForms) {
|
||||||
const score = await judgeRecommendation(buildAUQ(text));
|
const score = await judgeRecommendationPanel(buildAUQ(text));
|
||||||
expect(score.present, `[hedge:${label}] present should be true`).toBe(true);
|
expect(score.present, `[hedge:${label}] present should be true`).toBe(true);
|
||||||
expect(
|
expect(
|
||||||
score.commits,
|
score.commits,
|
||||||
|
|||||||
Reference in new issue
Block a user