mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
test(llm-judge): sample every judge as a pre-registered 3-sample panel
Each of the 24 skill-llm-eval judges now draws EVAL_POLICY.judge.samples independent samples of the same prompt concurrently inside the unchanged JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean against the unchanged threshold; booleans (would_browse, consistent) on a strict majority. An erroring sample fails the whole panel and is never resampled; a refusal is an unscored panel only when every sample refused. callJudge's 429 backoff stays: it is transport before any model output. The workflow-judge cache stores and validates only complete panels, and its identity now records the panel and zero file retries. Harness tests that pinned one provider call per case now pin the panel size.
This commit is contained in:
1 parent
adced7e046
commit
3f68572cab
7 files changed
+298
-112
No files matched your search
@@ -23,6 +23,8 @@ export interface JudgeScore {
|
||||
reasoning: string;
|
||||
}
|
||||
|
||||
export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const;
|
||||
|
||||
export interface JudgeRefusalEvidence {
|
||||
stop_reason: 'refusal';
|
||||
response_id: string | null;
|
||||
@@ -196,6 +198,63 @@ export async function callJudge<T>(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this
|
||||
* helper (imported by many paid tests) does not pull the quarantine registry
|
||||
* into their touchfile closure. test/judge-panel.test.ts pins the two equal.
|
||||
*/
|
||||
export const JUDGE_PANEL_SAMPLES = 3;
|
||||
|
||||
/**
|
||||
* Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of
|
||||
* independent samples of the SAME prompt concurrently, inside its unchanged
|
||||
* JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
|
||||
* against the unchanged minimum; boolean fields gate on a strict majority.
|
||||
* A sample that errors (refusal, truncation, non-JSON, malformed field) fails
|
||||
* the whole panel and is never resampled. callJudge's 429 backoff happens
|
||||
* before any model output exists, so it is transport, not a verdict retry.
|
||||
*/
|
||||
export async function judgePanel<T>(sample: () => Promise<T>): Promise<T[]> {
|
||||
const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample()));
|
||||
const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []);
|
||||
if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult<T>).value);
|
||||
const first = failures[0]!;
|
||||
// A refusal is an unscored panel only when EVERY sample refused; a partial
|
||||
// refusal beside scored samples is an ordinary failed panel.
|
||||
if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) {
|
||||
throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`);
|
||||
}
|
||||
throw first.reason;
|
||||
}
|
||||
|
||||
/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */
|
||||
export function judgePanelMean<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, keys: readonly K[]): Record<K, number> {
|
||||
if (samples.length === 0) throw new Error('Judge panel has no samples');
|
||||
return Object.fromEntries(keys.map(key => {
|
||||
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
|
||||
const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value));
|
||||
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`);
|
||||
return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length];
|
||||
})) as Record<K, number>;
|
||||
}
|
||||
|
||||
/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */
|
||||
export function judgePanelMajority<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, key: K): boolean {
|
||||
if (samples.length === 0) throw new Error('Judge panel has no samples');
|
||||
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
|
||||
const bad = values.findIndex(value => typeof value !== 'boolean');
|
||||
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`);
|
||||
return values.filter(value => value === true).length * 2 > values.length;
|
||||
}
|
||||
|
||||
/** Sample reasoning lines, numbered, for the collector record. */
|
||||
export function judgePanelReasoning(samples: ReadonlyArray<unknown>): string {
|
||||
return samples.map((sample, index) => {
|
||||
const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined;
|
||||
return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`;
|
||||
}).join('\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Score documentation quality on clarity/completeness/actionability (1-5).
|
||||
*/
|
||||
|
||||
@@ -72,7 +72,12 @@ export const CASE_CI_EXCLUDE: Record<string, { reason: string; tracking: string
|
||||
* new-policy trials; exit at >= `exit.rate` over >=
|
||||
* `exit.minTrials`; at most `capFraction` of each tier's
|
||||
* blocking cases; an entry expires after `expiryWeeklyRuns`.
|
||||
* drift - one-sided Fisher exact alarm between input-identity series.
|
||||
* judge - a judge case draws `samples` independent samples of one
|
||||
* prompt concurrently; numeric dimensions gate on the panel
|
||||
* mean against the unchanged threshold, booleans on a strict
|
||||
* majority; an erroring sample fails the panel, never resampled.
|
||||
* drift - one-sided Fisher exact alarm between input-identity series
|
||||
* (Holm-controlled across the cases tested in one report).
|
||||
* infraRedispatch - a census whose every red verdict is machine-classified
|
||||
* INFRA or INCOMPLETE may be re-dispatched this many times as
|
||||
* a new run; both runs are reported.
|
||||
@@ -86,6 +91,7 @@ export const EVAL_POLICY = {
|
||||
capFraction: 0.10,
|
||||
expiryWeeklyRuns: 8,
|
||||
},
|
||||
judge: { samples: 3 },
|
||||
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
|
||||
infraRedispatch: 1,
|
||||
} as const;
|
||||
|
||||
@@ -4,7 +4,7 @@ import * as path from 'node:path';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
|
||||
import { JUDGE_MS } from './eval-budgets';
|
||||
import type { JudgeScore } from './llm-judge';
|
||||
import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge';
|
||||
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input';
|
||||
import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache,
|
||||
type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache';
|
||||
@@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres
|
||||
|| typeof value.reasoning !== 'string'
|
||||
|| (structuredResponse && (!value.reasoning.trim()
|
||||
|| value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false;
|
||||
return (['clarity', 'completeness', 'actionability'] as const).every(key =>
|
||||
return JUDGE_SCORE_DIMENSIONS.every(key =>
|
||||
typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5);
|
||||
}
|
||||
|
||||
const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 };
|
||||
|
||||
/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */
|
||||
export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array<JudgeScore & EvalCacheValue> } {
|
||||
if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples'
|
||||
|| !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES
|
||||
|| !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false;
|
||||
const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS);
|
||||
return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]);
|
||||
}
|
||||
|
||||
export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null;
|
||||
lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null;
|
||||
/** The attempt guard is rechecked after synchronous input/provenance reads. */
|
||||
publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined;
|
||||
publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined;
|
||||
} {
|
||||
const env = opts.env ?? process.env;
|
||||
const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined };
|
||||
const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined };
|
||||
const pr = Number(env.EVALS_CACHE_PR);
|
||||
// Runtime ID is the immutable CI image manifest, not a mutable image tag.
|
||||
// Nonstandard Node/Bun preload code or custom model endpoints need a separate
|
||||
@@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
|
||||
prompts: { [opts.testName]: prompt },
|
||||
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
|
||||
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1,
|
||||
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0,
|
||||
panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' },
|
||||
...(opts.stream ? { stream: true } : {}),
|
||||
...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
|
||||
response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) },
|
||||
@@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
return {
|
||||
lookup() {
|
||||
const result = lookupEvalInputCache({ ...common, identity: before,
|
||||
validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) });
|
||||
validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) });
|
||||
return result.status === 'reused'
|
||||
? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null;
|
||||
? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null;
|
||||
},
|
||||
publish(scores, isActive = () => true) {
|
||||
publish(samples, isActive = () => true) {
|
||||
// Caller reaches here ONLY after its actual assertions passed. A later
|
||||
// failed case in the file does not erase this independently completed case.
|
||||
if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
|
||||
const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) };
|
||||
if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
|
||||
const after = currentIdentity();
|
||||
const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID;
|
||||
if (!after || !runId || !isActive()) return;
|
||||
@@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
cancelled: false, skipped: 0, failed: 0, passed: 1,
|
||||
cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }],
|
||||
source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() },
|
||||
result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning },
|
||||
result: panel,
|
||||
} });
|
||||
// A slow synchronous write can consume the recording allowance. The
|
||||
// caller withdraws this new receipt if its final deadline check fails.
|
||||
|
||||
Reference in new issue
Block a user