mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 09:56:57 +02:00
Merge remote-tracking branch 'origin/capy/rel-b' into capy/rel-a
This commit is contained in:
commit
a385e5de18
16 files changed
+1632
-234
No files matched your search
@@ -11,7 +11,7 @@ import { selectTests } from './helpers/test-selection';
|
||||
import { E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data';
|
||||
import { selectPrProfile } from '../scripts/test-pr-profile';
|
||||
import { JUDGE_MS } from './helpers/eval-budgets';
|
||||
import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
|
||||
import { JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge';
|
||||
import { COOKIE_MANUAL_REVIEW_FILE, getCookieWorkflowManualReview, isManualReviewEntry } from './helpers/cookie-workflow-manual-review';
|
||||
|
||||
const ROOT = resolve(import.meta.dir, '..');
|
||||
@@ -70,7 +70,7 @@ function actualCookieCallback(root: string, overrides: {
|
||||
const records: EvalTestEntry[] = [];
|
||||
const attempts = new Map<string, { attempt: number }>();
|
||||
let callback: () => Promise<void> = async () => { throw new Error('Judge callback was not registered'); };
|
||||
new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', registration)(
|
||||
new Function('describeIfSelected', 'testIfSelected', 'ROOT', 'buildCookieWorkflowJudgeInput', 'resolveEvalModel', 'callJudge', 'COOKIE_WORKFLOW_JUDGE', 'JUDGE_MS', 'WORKFLOW_JUDGE_TEST_MS', 'WORKFLOW_JUDGE_RECORD_MS', 'evalCollector', 'expect', 'console', 'readWorkflowJudgeInput', 'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'performance', 'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore', 'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS', registration)(
|
||||
(_suite: string, names: string[], run: () => void) => { expect(names).toEqual([NAME]); run(); },
|
||||
(name: string, run: () => Promise<void>, budget: number) => { expect(name).toBe(NAME); expect(budget).toBe(JUDGE_MS + 10_000); callback = run; },
|
||||
root, buildCookieWorkflowJudgeInput, (_kind: string, explicit?: string) => explicit ?? 'fixture-model',
|
||||
@@ -85,7 +85,7 @@ function actualCookieCallback(root: string, overrides: {
|
||||
attempts, overrides.clock ? { now: overrides.clock } : performance,
|
||||
overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout,
|
||||
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS,
|
||||
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore,
|
||||
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore, judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS,
|
||||
);
|
||||
return { run: () => callback(), requests, records, attempts };
|
||||
}
|
||||
@@ -96,7 +96,7 @@ describe('cookie workflow judge input', () => {
|
||||
approveFixture(root);
|
||||
const h = actualCookieCallback(root, { judge: async () => { throw refusal(); } });
|
||||
await h.run();
|
||||
expect(h.requests).toHaveLength(1);
|
||||
expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
|
||||
expect(h.records).toHaveLength(1);
|
||||
expect(h.records[0]).toMatchObject({ passed: false, execution: 'executed', exit_reason: 'provider_refusal' });
|
||||
expect(isManualReviewEntry(h.records[0])).toBe(true);
|
||||
@@ -161,7 +161,7 @@ describe('cookie workflow judge input', () => {
|
||||
const root = fixture(); approveFixture(root);
|
||||
let calls = 0;
|
||||
const h = actualCookieCallback(root, { judge: async () => {
|
||||
if (++calls === 1) return { ...passingScore, clarity: 1 };
|
||||
if (++calls <= JUDGE_PANEL_SAMPLES) return { ...passingScore, clarity: 1 };
|
||||
throw refusal();
|
||||
} });
|
||||
await expect(h.run()).rejects.toThrow();
|
||||
@@ -275,7 +275,7 @@ describe('cookie workflow judge input', () => {
|
||||
let scores = passingScore;
|
||||
const h = actualCookieCallback(root, { judge: async () => scores });
|
||||
await h.run();
|
||||
expect(h.requests).toHaveLength(1);
|
||||
expect(h.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
|
||||
expect(h.requests[0].prompt).toBe(input.prompt);
|
||||
expect(h.requests[0].model).toBe(COOKIE_WORKFLOW_JUDGE.model);
|
||||
expect(h.requests[0].signal).toBeInstanceOf(AbortSignal);
|
||||
@@ -283,7 +283,7 @@ describe('cookie workflow judge input', () => {
|
||||
expect(existsSync(join(root, 'cache'))).toBe(false);
|
||||
const fresh = actualCookieCallback(root);
|
||||
await fresh.run();
|
||||
expect(fresh.requests).toHaveLength(1);
|
||||
expect(fresh.requests).toHaveLength(JUDGE_PANEL_SAMPLES);
|
||||
for (const dimension of ['clarity', 'completeness', 'actionability'] as const) {
|
||||
scores = { ...COOKIE_WORKFLOW_JUDGE.thresholds, [dimension]: COOKIE_WORKFLOW_JUDGE.thresholds[dimension] - 1, reasoning: 'Synthetic failing fixture score' };
|
||||
await expect(h.run()).rejects.toThrow();
|
||||
|
||||
@@ -13,6 +13,13 @@ import * as path from 'node:path';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { aggregate, collectEvalFiles } from '../scripts/eval-flake-rank';
|
||||
import { manualReviewFixture } from './helpers/manual-judge-review-fixture';
|
||||
import {
|
||||
analyzePassRates, attributeLegacyRecord, backfillEvalFiles, caseSeriesIdentities, downloadRunArtifacts, fisherOneSidedLower,
|
||||
formatPassRates, holmRejections, listWeeklyRuns, quarantinePolicyProblems, quarantineRunsSince, readTrialOutcomeDir,
|
||||
wilsonInterval, type HistoryFetcher, type PassRatePolicy, type QuarantineEntry, type Registry, type TrialRecord,
|
||||
} from '../scripts/eval-flake-rank';
|
||||
import { EVAL_POLICY } from './helpers/periodic-exclude-data';
|
||||
import { TRIAL_OUTCOME_SCHEMA, formatTrialOutcomes } from './helpers/eval-store';
|
||||
|
||||
const entry = (name: string, passed: boolean, attempt: number) => ({
|
||||
name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1,
|
||||
@@ -39,9 +46,10 @@ describe('eval-flake-rank aggregate', () => {
|
||||
const display = spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), '--dir', dir],
|
||||
{ encoding: 'utf8', timeout: 10_000 });
|
||||
expect(display.status, display.stderr).toBe(0);
|
||||
expect(display.stdout).toContain('fails/runs manual');
|
||||
expect(display.stdout).toContain('0/1');
|
||||
expect(display.stdout).toContain(manual.name);
|
||||
// pass-rates view: the prior automated pass is the one scored pre-policy
|
||||
// trial; the manual acceptance is counted in its own column, never scored.
|
||||
expect(display.stdout).toContain('pre-policy manual case');
|
||||
expect(display.stdout).toMatch(new RegExp(`1/1 \\[[^\\]]+\\]\\s+1 ${manual.name.replace(/[.*+?^${}()|[\]\\/]/g, '\\$&')}`));
|
||||
fs.writeFileSync(path.join(dir, 'invalid-retry.json'), run([
|
||||
{ ...ordinary, attempt: 1 }, { ...manual, attempt: 2 },
|
||||
]));
|
||||
@@ -97,3 +105,300 @@ describe('eval-flake-rank aggregate', () => {
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
});
|
||||
|
||||
// --- pass-rates ---
|
||||
|
||||
const registry: Registry = {
|
||||
kinds: { 'rule-a': 'rule', 'beh-b': 'behavior', 'gate-c': 'rule', 'mar-d': 'rule', 'judge one': 'judge',
|
||||
...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, 'rule'])) },
|
||||
tiers: { 'rule-a': 'periodic', 'beh-b': 'periodic', 'gate-c': 'gate', 'mar-d': 'marathon',
|
||||
...Object.fromEntries(Array.from({ length: 18 }, (_, i) => [`filler-${i}`, i < 9 ? 'gate' : 'periodic'])) },
|
||||
touchfiles: { 'rule-a': ['test/skill-e2e-a.test.ts', 'a/**'], 'beh-b': ['test/skill-e2e-b.test.ts', 'b/**'],
|
||||
'gate-c': ['test/skill-e2e-shared.test.ts'], 'mar-d': ['test/skill-e2e-shared.test.ts'] },
|
||||
judgeTouchfiles: { 'judge one': ['j/SKILL.md'] },
|
||||
globals: ['harness/**'],
|
||||
testNames: { 'gate-c': '/gate c labeled' },
|
||||
};
|
||||
|
||||
let clock = 0;
|
||||
function trial(id: string, outcome: 'passed' | 'failed' | 'skipped', extra: Partial<TrialRecord> = {}): TrialRecord {
|
||||
clock += 1;
|
||||
return {
|
||||
schema: TRIAL_OUTCOME_SCHEMA, case: id, file: 'test/x.test.ts', tier: registry.tiers[id] ?? 'judge',
|
||||
kind: registry.kinds[id]!, trial: 1, panel: { n: 1, k: 1 }, attempt: 1, outcome,
|
||||
...(outcome === 'failed' ? { failure_class: 'assertion' as const } : {}),
|
||||
duration_ms: 1, cost_usd: 0, model: 'model-x', cli_version: '2.1.284', policy_version: 1, quarantined: false,
|
||||
execution: 'executed', source: 'shard', run_id: `run-${clock}`, recorded_at: new Date(Date.UTC(2026, 9, 1) + clock * 60_000).toISOString(),
|
||||
series_identity: 'id-1', ...extra,
|
||||
};
|
||||
}
|
||||
const many = (id: string, passes: number, fails: number, extra: Partial<TrialRecord> = {}) =>
|
||||
[...Array.from({ length: passes }, () => trial(id, 'passed', extra)), ...Array.from({ length: fails }, () => trial(id, 'failed', extra))];
|
||||
const analyze = (records: TrialRecord[], quarantine: Record<string, QuarantineEntry> = {}, extra = {}) =>
|
||||
analyzePassRates(records, { registry, quarantine, now: Date.UTC(2026, 9, 2), ...extra });
|
||||
const qEntry = (overrides: Partial<QuarantineEntry> = {}): QuarantineEntry => ({
|
||||
reason: 'Detector graded the posture wording; 7 of 10 fresh trials failed only the regex, transcripts attached.',
|
||||
failureClass: 'detector', tracking: 'TODOS.md "x"', owner: 'garrytan', enteredAt: '2026-09-29',
|
||||
exit: '>= 97% over >= 10 trials on the current identity', ...overrides,
|
||||
});
|
||||
|
||||
describe('pass-rates statistics', () => {
|
||||
test('Wilson bounds match the documented policy arithmetic', () => {
|
||||
expect(wilsonInterval(10, 10).lo).toBeCloseTo(0.7225, 4);
|
||||
expect(wilsonInterval(6, 6).lo).toBeCloseTo(0.6097, 4);
|
||||
expect(wilsonInterval(125, 125).lo).toBeCloseTo(0.9702, 4);
|
||||
expect(wilsonInterval(10, 10).hi).toBe(1);
|
||||
expect(wilsonInterval(0, 0)).toEqual({ lo: 0, hi: 1 });
|
||||
const mid = wilsonInterval(7, 10);
|
||||
expect(mid.lo).toBeGreaterThan(0.39); expect(mid.hi).toBeLessThan(0.9);
|
||||
});
|
||||
|
||||
test('one-sided Fisher exact matches a known table and is one-sided', () => {
|
||||
expect(fisherOneSidedLower(4, 6, 6, 6)).toBeCloseTo(0.227272727, 8);
|
||||
expect(fisherOneSidedLower(6, 6, 4, 6)).toBe(1);
|
||||
expect(fisherOneSidedLower(0, 10, 10, 10)).toBeLessThan(1e-4);
|
||||
});
|
||||
|
||||
test('Holm rejects step-down and stops at the first non-rejection', () => {
|
||||
expect([...holmRejections([0.001, 0.02, 0.04], 0.05)].sort()).toEqual([0, 1, 2]);
|
||||
expect([...holmRejections([0.001, 0.03, 0.04], 0.05)]).toEqual([0]);
|
||||
expect([...holmRejections([0.03, 0.04], 0.05)]).toEqual([]);
|
||||
expect([...holmRejections([], 0.05)]).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('pass-rates labels', () => {
|
||||
test('thin history is INCONCLUSIVE, and after this PR every series starts there', () => {
|
||||
const report = analyze(many('rule-a', 9, 0));
|
||||
expect(report.cases[0]).toMatchObject({ case: 'rule-a', label: 'INCONCLUSIVE' });
|
||||
expect(formatPassRates(analyze([]))).toContain('every series starts INCONCLUSIVE');
|
||||
});
|
||||
|
||||
test('PASSING, FLAKY and FAILING come from the interval against the entry rate', () => {
|
||||
expect(analyze(many('rule-a', 12, 0)).cases[0]!.label).toBe('PASSING');
|
||||
expect(analyze(many('beh-b', 10, 1)).cases[0]!.label).toBe('FLAKY');
|
||||
expect(analyze(many('beh-b', 2, 10)).cases[0]!.label).toBe('FAILING');
|
||||
});
|
||||
|
||||
test('BROKEN: the latest run is 0/n after a prior interval at or above the entry rate', () => {
|
||||
const prior = many('beh-b', 80, 0, { run_id: 'old' });
|
||||
const latest = [1, 2, 3].map(n => trial('beh-b', 'failed', { run_id: 'new', trial: n, panel: { n: 3, k: 2 } }));
|
||||
expect(analyze([...prior, ...latest]).cases[0]!.label).toBe('BROKEN');
|
||||
});
|
||||
|
||||
test('skipped trials carry no verdict; infra failures count as failed trials', () => {
|
||||
const stats = analyze([...many('rule-a', 10, 0), trial('rule-a', 'skipped'),
|
||||
trial('rule-a', 'failed', { failure_class: 'infra' })]).cases[0]!.current!;
|
||||
expect(stats).toMatchObject({ passes: 10, trials: 11, infra: 1 });
|
||||
});
|
||||
|
||||
test('a new identity, model or CLI starts a new series; earlier series stay visible', () => {
|
||||
const report = analyze([...many('rule-a', 10, 0), ...many('rule-a', 3, 0, { series_identity: 'id-2' }),
|
||||
...many('rule-a', 2, 0, { series_identity: 'id-2', cli_version: '2.1.285' })]);
|
||||
const c = report.cases[0]!;
|
||||
expect(c.series).toHaveLength(3);
|
||||
expect(c.current).toMatchObject({ identity: 'id-2', cli: '2.1.285', trials: 2 });
|
||||
expect(c.previous).toMatchObject({ identity: 'id-2', cli: '2.1.284', trials: 3 });
|
||||
expect(c.label).toBe('INCONCLUSIVE');
|
||||
});
|
||||
});
|
||||
|
||||
describe('pass-rates alarms count post-policy trials of the current series only', () => {
|
||||
test('backfilled pre-policy failures are displayed but never alarm', () => {
|
||||
const report = analyze(many('rule-a', 2, 20, { policy_version: 0, source: 'backfill' }));
|
||||
expect(report.alarms).toEqual([]);
|
||||
expect(report.cases[0]!.prePolicy).toMatchObject({ passes: 2, trials: 22 });
|
||||
expect(report.cases[0]!.label).toBe('INCONCLUSIVE');
|
||||
});
|
||||
|
||||
test('drift proposes quarantine for a blocking case below the entry rule; a rule case is flagged as behaving like behavior', () => {
|
||||
const kinds = analyze([...many('rule-a', 8, 2), ...many('beh-b', 8, 2), ...many('mar-d', 0, 10)]).alarms.map(a => `${a.kind}:${a.case}`);
|
||||
expect([...kinds].sort()).toEqual(['drift:beh-b', 'drift:rule-a', 'rule-as-behavior:mar-d', 'rule-as-behavior:rule-a']);
|
||||
expect(analyze(many('rule-a', 19, 1)).alarms).toEqual([]);
|
||||
});
|
||||
|
||||
test('the Fisher regression alarm needs the minimum trials on both sides', () => {
|
||||
const old = many('gate-c', 6, 0, { series_identity: 'old' });
|
||||
const fresh = many('gate-c', 0, 6, { series_identity: 'new' });
|
||||
expect(analyze([...old, ...fresh]).alarms.map(a => a.kind)).toContain('regression');
|
||||
expect(analyze([...old, ...fresh.slice(0, 5)]).alarms.map(a => a.kind)).not.toContain('regression');
|
||||
});
|
||||
|
||||
test('quarantine exit, expiry and cap', () => {
|
||||
const exit = analyze(many('beh-b', 10, 0), { 'beh-b': qEntry() }).alarms.map(a => a.kind);
|
||||
expect(exit).toContain('quarantine-exit');
|
||||
expect(exit).not.toContain('drift');
|
||||
const weekly = Array.from({ length: 8 }, (_, i) => new Date(Date.UTC(2026, 8, 30) + i * 7 * 86_400_000).toISOString());
|
||||
expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly }).alarms.map(a => a.kind)).toContain('quarantine-expired');
|
||||
expect(analyze([], { 'beh-b': qEntry() }, { weeklyRuns: weekly.slice(0, 7) }).alarms.map(a => a.kind)).not.toContain('quarantine-expired');
|
||||
expect(quarantineRunsSince('2026-09-01', undefined, Date.UTC(2026, 9, 27))).toBe(8);
|
||||
expect(quarantineRunsSince('not a date', undefined, 0)).toBe(Number.POSITIVE_INFINITY);
|
||||
});
|
||||
});
|
||||
|
||||
describe('quarantine policy', () => {
|
||||
const policy: PassRatePolicy = EVAL_POLICY;
|
||||
const now = Date.UTC(2026, 9, 2);
|
||||
test('a valid entry has no problems', () => {
|
||||
expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]);
|
||||
});
|
||||
|
||||
test('a product defect, a missing diagnosis or field, a bad date or a non-blocking case is rejected', () => {
|
||||
const problems = (quarantine: Record<string, QuarantineEntry>) => quarantinePolicyProblems(quarantine, registry, policy, now).map(p => p.message);
|
||||
expect(problems({ 'beh-b': qEntry({ failureClass: 'product' as QuarantineEntry['failureClass'] }) }).join()).toContain('never quarantined');
|
||||
expect(problems({ 'beh-b': qEntry({ reason: 'flaky' }) }).join()).toContain('written diagnosis');
|
||||
expect(problems({ 'beh-b': qEntry({ owner: ' ' }) }).join()).toContain('missing owner');
|
||||
expect(problems({ 'beh-b': qEntry({ enteredAt: '09/29/2026' }) }).join()).toContain('YYYY-MM-DD');
|
||||
expect(problems({ 'beh-b': qEntry({ enteredAt: '2027-01-01' }) }).join()).toContain('future');
|
||||
expect(problems({ 'mar-d': qEntry() }).join()).toContain('not blocking');
|
||||
expect(problems({ 'judge one': qEntry() }).join()).toContain('no registered E2E case');
|
||||
expect(problems({ ghost: qEntry() }).join()).toContain('no registered E2E case');
|
||||
});
|
||||
|
||||
test('at most 10% of a tier may be quarantined', () => {
|
||||
// 11 periodic cases in the fixture registry: the cap is 1.
|
||||
expect(quarantinePolicyProblems({ 'beh-b': qEntry() }, registry, policy, now)).toEqual([]);
|
||||
const over = quarantinePolicyProblems({ 'beh-b': qEntry(), 'rule-a': qEntry() }, registry, policy, now);
|
||||
expect(over.map(p => p.kind)).toEqual(['quarantine-cap']);
|
||||
expect(over[0]!.message).toContain('2 quarantined periodic cases exceed the 10% cap (1 of 11)');
|
||||
});
|
||||
});
|
||||
|
||||
describe('pass-rates inputs', () => {
|
||||
test('trial-outcomes JSONL is schema-validated; invalid lines are reported, never guessed', () => {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-'));
|
||||
const valid = trial('rule-a', 'passed');
|
||||
fs.mkdirSync(path.join(dir, 'nested'));
|
||||
fs.writeFileSync(path.join(dir, 'nested', 'trial-outcomes.jsonl'), formatTrialOutcomes([valid]) + '{"schema":"other"}\nnot json\n');
|
||||
fs.writeFileSync(path.join(dir, 'unrelated.jsonl'), formatTrialOutcomes([valid]));
|
||||
const read = readTrialOutcomeDir(dir);
|
||||
expect(read.records).toHaveLength(1);
|
||||
expect(read.records[0]).toMatchObject({ case: 'rule-a', series_identity: 'id-1' });
|
||||
expect(read.errors).toHaveLength(2);
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('legacy records attribute by shard suffix, id, label or single-owner file, else stay unattributed', () => {
|
||||
expect(attributeLegacyRecord('/anything', 'skill-e2e-b--beh-b', registry)).toBe('beh-b');
|
||||
expect(attributeLegacyRecord('rule-a', 'skill-e2e-zzz', registry)).toBe('rule-a');
|
||||
expect(attributeLegacyRecord('/Rule a', 'skill-e2e-zzz', registry)).toBe('rule-a');
|
||||
expect(attributeLegacyRecord('/rule a extra', 'skill-e2e-zzz', registry)).toBeNull();
|
||||
expect(attributeLegacyRecord('/gate c labeled', undefined, registry)).toBe('gate-c');
|
||||
expect(attributeLegacyRecord('/a display name', 'skill-e2e-a', registry)).toBe('rule-a');
|
||||
expect(attributeLegacyRecord('/shared display', 'skill-e2e-shared', registry)).toBeNull();
|
||||
});
|
||||
|
||||
test('backfill keeps only first attempts, defaults a missing attempt to 1, and labels records pre-policy', () => {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-backfill-'));
|
||||
fs.writeFileSync(path.join(dir, 'run.json'), run([
|
||||
{ ...entry_('rule-a', false, 1), exit_reason: 'timeout' }, entry_('rule-a', true, 2),
|
||||
{ name: 'beh-b', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 },
|
||||
entry_('/unknown display', true, 1),
|
||||
], { shard: 'skill-e2e-zzz' }));
|
||||
const { records, unattributed } = backfillEvalFiles(collectEvalFiles(dir), { run_id: '42', sha: 'abc' }, registry);
|
||||
expect(records.map(r => [r.case, r.outcome, r.failure_class, r.policy_version, r.source, r.run_id]))
|
||||
.toEqual([['rule-a', 'failed', 'timeout', 0, 'backfill', '42'], ['beh-b', 'passed', undefined, 0, 'backfill', '42']]);
|
||||
expect(formatTrialOutcomes(records)).toContain(TRIAL_OUTCOME_SCHEMA);
|
||||
expect(unattributed).toEqual(['/unknown display']);
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('series identity follows the case\'s own touchfiles, not GLOBAL_TOUCHFILES', () => {
|
||||
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-series-'));
|
||||
const git = (...args: string[]) => spawnSync('git', args, { cwd: root, encoding: 'utf8', timeout: 10_000 });
|
||||
for (const [file, body] of [['a/x.ts', '1'], ['b/y.ts', '1'], ['harness/run.ts', '1'], ['test/skill-e2e-a.test.ts', '1']] as const) {
|
||||
fs.mkdirSync(path.join(root, path.dirname(file)), { recursive: true });
|
||||
fs.writeFileSync(path.join(root, file), body);
|
||||
}
|
||||
const snapshot = () => { expect(git('add', '-A').status).toBe(0); return caseSeriesIdentities(['rule-a', 'beh-b'], root, registry); };
|
||||
expect(git('init', '-q').status).toBe(0);
|
||||
const first = snapshot();
|
||||
expect(first['rule-a']).not.toBe(first['beh-b']);
|
||||
fs.writeFileSync(path.join(root, 'harness/run.ts'), '2');
|
||||
expect(snapshot()).toEqual(first);
|
||||
fs.writeFileSync(path.join(root, 'a/x.ts'), '2');
|
||||
const next = snapshot();
|
||||
expect(next['rule-a']).not.toBe(first['rule-a']);
|
||||
expect(next['beh-b']).toBe(first['beh-b']);
|
||||
fs.rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
});
|
||||
|
||||
describe('pass-rates history fetch (injected, no network)', () => {
|
||||
function storedZip(files: Record<string, string>): Buffer {
|
||||
const locals: Buffer[] = [], centrals: Buffer[] = [];
|
||||
let offset = 0;
|
||||
for (const [name, text] of Object.entries(files)) {
|
||||
const data = Buffer.from(text), fileName = Buffer.from(name), crc = Bun.hash.crc32(data) >>> 0;
|
||||
const local = Buffer.alloc(30); local.writeUInt32LE(0x04034b50, 0); local.writeUInt16LE(20, 4);
|
||||
local.writeUInt32LE(crc, 14); local.writeUInt32LE(data.length, 18); local.writeUInt32LE(data.length, 22); local.writeUInt16LE(fileName.length, 26);
|
||||
const central = Buffer.alloc(46); central.writeUInt32LE(0x02014b50, 0); central.writeUInt16LE(20, 4); central.writeUInt16LE(20, 6);
|
||||
central.writeUInt32LE(crc, 16); central.writeUInt32LE(data.length, 20); central.writeUInt32LE(data.length, 24);
|
||||
central.writeUInt16LE(fileName.length, 28); central.writeUInt32LE(offset, 42);
|
||||
locals.push(local, fileName, data); centrals.push(central, fileName);
|
||||
offset += 30 + fileName.length + data.length;
|
||||
}
|
||||
const size = centrals.reduce((sum, b) => sum + b.length, 0);
|
||||
const end = Buffer.alloc(22); end.writeUInt32LE(0x06054b50, 0); end.writeUInt16LE(Object.keys(files).length, 8);
|
||||
end.writeUInt16LE(Object.keys(files).length, 10); end.writeUInt32LE(size, 12); end.writeUInt32LE(offset, 16);
|
||||
return Buffer.concat([...locals, ...centrals, end]);
|
||||
}
|
||||
|
||||
test('lists runs per branch, deduplicated and newest first', () => {
|
||||
const fetcher: HistoryFetcher = {
|
||||
listRuns: (_repo, _workflow, branch) => branch === 'main'
|
||||
? [{ id: 1, attempt: 1, sha: 'a', branch, createdAt: '2026-09-01T00:00:00Z' }, { id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }]
|
||||
: [{ id: 3, attempt: 1, sha: 'c', branch, createdAt: '2026-09-15T00:00:00Z' }, { id: 2, attempt: 2, sha: 'b', branch, createdAt: '2026-09-08T00:00:00Z' }],
|
||||
listArtifacts: () => [], downloadZip: () => { throw new Error('unused'); },
|
||||
};
|
||||
expect(listWeeklyRuns({ repo: 'o/r', workflow: 'evals-periodic.yml', branches: ['feature', 'main'], limit: 10, fetcher }).map(r => r.id)).toEqual([3, 2, 1]);
|
||||
});
|
||||
|
||||
test('downloads only matching, bounded artifacts once, and caches them', () => {
|
||||
const cacheDir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cache-'));
|
||||
const downloads: number[] = [];
|
||||
const fetcher: HistoryFetcher = {
|
||||
listRuns: () => [],
|
||||
listArtifacts: () => [{ id: 10, name: 'trial-outcomes-gate', size: 100 }, { id: 11, name: 'paid-slice-1', size: 100 },
|
||||
{ id: 12, name: 'trial-outcomes-huge', size: 10 ** 9 }, { id: 13, name: 'trial-outcomes/../escape', size: 1 }],
|
||||
downloadZip: (_repo, id, destination) => { downloads.push(id); fs.writeFileSync(destination, storedZip({ 'trial-outcomes.jsonl': formatTrialOutcomes([trial('rule-a', 'passed')]) })); },
|
||||
};
|
||||
const options = { repo: 'o/r', run: { id: 7, attempt: 1, sha: 's', branch: 'main', createdAt: '' }, cacheDir, fetcher,
|
||||
match: (name: string) => name.startsWith('trial-outcomes') };
|
||||
const dirs = downloadRunArtifacts(options);
|
||||
expect(downloads).toEqual([10]);
|
||||
expect(dirs).toHaveLength(1);
|
||||
expect(readTrialOutcomeDir(dirs[0]!).records.map(r => r.case)).toEqual(['rule-a']);
|
||||
expect(downloadRunArtifacts(options)).toEqual(dirs);
|
||||
expect(downloads).toEqual([10]);
|
||||
fs.rmSync(cacheDir, { recursive: true, force: true });
|
||||
});
|
||||
});
|
||||
|
||||
describe('pass-rates CLI', () => {
|
||||
const cli = (args: string[]) => spawnSync(process.execPath, [path.resolve(import.meta.dir, '../scripts/eval-flake-rank.ts'), ...args],
|
||||
{ encoding: 'utf8', timeout: 20_000 });
|
||||
|
||||
test('--dir prints per-case pass rates; --gate fails only on ACTION REQUIRED', () => {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'passrates-cli-'));
|
||||
const id = 'plan-ceo-review-format-mode';
|
||||
const records = Array.from({ length: 12 }, (_, i) => ({ ...trial(id, i < 11 ? 'passed' : 'failed'), kind: 'behavior' as const, tier: 'periodic' }));
|
||||
fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records));
|
||||
const shown = cli(['--dir', dir, '--case', id]);
|
||||
expect(shown.status, shown.stderr).toBe(0);
|
||||
expect(shown.stdout).toContain(`11/12 [`);
|
||||
expect(shown.stdout).toMatch(new RegExp(`FLAKY\\s+behavior\\s+periodic.*${id}`));
|
||||
expect(shown.stdout).toContain('ACTION REQUIRED');
|
||||
expect(shown.stdout).toContain(`[drift] ${id} passes 11/12`);
|
||||
expect(cli(['--dir', dir, '--gate']).status).toBe(1);
|
||||
fs.writeFileSync(path.join(dir, 'trial-outcomes.jsonl'), formatTrialOutcomes(records.slice(0, 11)));
|
||||
const clean = cli(['--dir', dir, '--gate', '--json']);
|
||||
expect(clean.status, clean.stdout).toBe(0);
|
||||
expect(JSON.parse(clean.stdout).cases[0]).toMatchObject({ case: id, label: 'PASSING', current: { passes: 11, trials: 11 } });
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
});
|
||||
|
||||
function entry_(name: string, passed: boolean, attempt: number) {
|
||||
return { name, suite: 's', tier: 'e2e', passed, attempt, duration_ms: 1000, cost_usd: 0.1 };
|
||||
}
|
||||
@@ -0,0 +1,107 @@
|
||||
/**
|
||||
* Eval kind registry (E2E_KINDS / BEHAVIOR_WHY in touchfiles-data.ts). The
|
||||
* kind fixes a case's trial policy before the run, so the registry must cover
|
||||
* every live case exactly once, every behavior case must name its tolerated
|
||||
* deviation, and a behavior case must be isolatable as its own trial shard.
|
||||
* A kind edit must re-select the case in the PR lane (map-diff).
|
||||
*/
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
|
||||
import { BEHAVIOR_WHY, E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data';
|
||||
import { diffTouchfileMapsCore, type TouchfileMaps } from './helpers/test-selection';
|
||||
import { CASE_TEST_NAMES, fileCaseRegistration } from '../scripts/test-paid-shards';
|
||||
import { isPaidTestFile } from './helpers/paid-test-set';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const KIND_RULE = "Pick the kind by what can make the verdict differ between two runs of the same commit: 'rule' when nothing "
|
||||
+ "stochastic decides it or it checks a contract the product must meet every run (the default); 'behavior' when a live "
|
||||
+ "model choice decides it and a sub-100% per-trial rate is acceptable (add a BEHAVIOR_WHY line); 'judge' when the only "
|
||||
+ 'stochastic step is an LLM judge scoring a fixed input.';
|
||||
|
||||
const liveIds = [...Object.keys(E2E_TIERS), ...Object.keys(LLM_JUDGE_TOUCHFILES)];
|
||||
const behaviorIds = Object.keys(E2E_KINDS).filter(id => E2E_KINDS[id] === 'behavior').sort();
|
||||
|
||||
describe('E2E_KINDS registry', () => {
|
||||
test('every live case has exactly one kind and no kind names a dead case', () => {
|
||||
const missing = liveIds.filter(id => !(id in E2E_KINDS));
|
||||
expect(missing.length, missing.length ? `add to E2E_KINDS:\n${missing.map(id => ` '${id}': 'rule', // <reason>`).join('\n')}\n${KIND_RULE}` : '').toBe(0);
|
||||
const unknown = Object.keys(E2E_KINDS).filter(id => !liveIds.includes(id));
|
||||
expect(unknown, `E2E_KINDS names ids that are neither E2E_TIERS nor LLM_JUDGE_TOUCHFILES keys`).toEqual([]);
|
||||
expect(new Set(liveIds).size).toBe(liveIds.length);
|
||||
});
|
||||
|
||||
test('kinds are rule, behavior or judge; every LLM-judge entry is judge-kind', () => {
|
||||
for (const [id, kind] of Object.entries(E2E_KINDS)) expect(['rule', 'behavior', 'judge'], id).toContain(kind);
|
||||
for (const id of Object.keys(LLM_JUDGE_TOUCHFILES)) expect(E2E_KINDS[id], `${id}: a workflow judge scores a fixed input`).toBe('judge');
|
||||
});
|
||||
|
||||
test('BEHAVIOR_WHY names the tolerance of exactly the behavior cases', () => {
|
||||
expect(Object.keys(BEHAVIOR_WHY).sort()).toEqual(behaviorIds);
|
||||
for (const id of behaviorIds) {
|
||||
expect(BEHAVIOR_WHY[id]!.trim().length, `${id}: BEHAVIOR_WHY must say why an occasional deviation is acceptable`).toBeGreaterThanOrEqual(30);
|
||||
}
|
||||
});
|
||||
|
||||
test('a behavior case is an isolatable trial shard: known literal registration and an exact Bun test name', () => {
|
||||
for (const id of behaviorIds) {
|
||||
const files = E2E_TOUCHFILES[id]!.filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file));
|
||||
expect(files.length, `${id}: no paid test file registers it`).toBeGreaterThan(0);
|
||||
for (const file of files) {
|
||||
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
|
||||
expect(fileCaseRegistration(file, source).known, `${id}: ${file} has a computed registration; behavior needs a literal one`).toBe(true);
|
||||
const name = CASE_TEST_NAMES[id] ?? id;
|
||||
const literal = new RegExp(`\\b(?:test(?:\\.serial|\\.concurrent)?|testIfSelected|testConcurrentIfSelected)\\(\\s*(['"\`])${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\1`);
|
||||
expect(literal.test(source), `${id}: ${file} must register the Bun test named '${name}'`).toBe(true);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test('the classification is the reviewed one: rule by default, 22 behavior, 25 judge', () => {
|
||||
const counts = Object.values(E2E_KINDS).reduce<Record<string, number>>((acc, kind) => ({ ...acc, [kind]: (acc[kind] ?? 0) + 1 }), {});
|
||||
expect(counts).toEqual({ rule: liveIds.length - 22 - 25, behavior: 22, judge: 25 });
|
||||
// Contract-shaped cases stay rule: ask-before-decide, plan-mode no-writes,
|
||||
// mandated steps, secrets, and the batching floor never ride a majority.
|
||||
for (const id of ['plan-ceo-mode-routing', 'plan-eng-multi-finding-batching', 'plan-design-review-plan-mode',
|
||||
'plan-eng-review-plan-mode', 'plan-ceo-section-loading', 'setup-gbrain-bad-token', 'qa-only-no-fix', 'review-sql-injection']) {
|
||||
expect(E2E_KINDS[id], id).toBe('rule');
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('kind edits re-select their case (map-diff)', () => {
|
||||
const base = (): TouchfileMaps => ({
|
||||
E2E_TOUCHFILES: { alpha: ['a/**'], beta: ['b/**'] },
|
||||
E2E_TIERS: { alpha: 'gate', beta: 'periodic' },
|
||||
LLM_JUDGE_TOUCHFILES: { 'judge one': ['j/SKILL.md'] },
|
||||
GLOBAL_TOUCHFILES: [],
|
||||
E2E_KINDS: { alpha: 'rule', beta: 'rule', 'judge one': 'judge' },
|
||||
BEHAVIOR_WHY: {},
|
||||
});
|
||||
|
||||
test('a rule -> behavior flip selects exactly that case', () => {
|
||||
const next = base();
|
||||
next.E2E_KINDS = { ...next.E2E_KINDS, beta: 'behavior' };
|
||||
next.BEHAVIOR_WHY = { beta: 'tolerated deviation' };
|
||||
expect(diffTouchfileMapsCore(base(), next).changedTests).toEqual(['beta']);
|
||||
});
|
||||
|
||||
test('a BEHAVIOR_WHY edit alone selects its case', () => {
|
||||
const old = base(); old.E2E_KINDS!.beta = 'behavior'; old.BEHAVIOR_WHY = { beta: 'one' };
|
||||
const next = base(); next.E2E_KINDS!.beta = 'behavior'; next.BEHAVIOR_WHY = { beta: 'two' };
|
||||
expect(diffTouchfileMapsCore(old, next).changedTests).toEqual(['beta']);
|
||||
});
|
||||
|
||||
test('a base revision without the kind maps selects every key', () => {
|
||||
const old = base(); delete old.E2E_KINDS; delete old.BEHAVIOR_WHY;
|
||||
expect(diffTouchfileMapsCore(old, base()).changedTests).toEqual(['alpha', 'beta', 'judge one']);
|
||||
});
|
||||
|
||||
test('dropping a kind entry while the case lives on counts as changed, not removed', () => {
|
||||
const next = base(); delete next.E2E_KINDS!.alpha;
|
||||
const result = diffTouchfileMapsCore(base(), next);
|
||||
expect(result.changedTests).toEqual(['alpha']);
|
||||
expect(result.removedTests).toEqual([]);
|
||||
});
|
||||
});
|
||||
@@ -23,6 +23,8 @@ export interface JudgeScore {
|
||||
reasoning: string;
|
||||
}
|
||||
|
||||
export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const;
|
||||
|
||||
export interface JudgeRefusalEvidence {
|
||||
stop_reason: 'refusal';
|
||||
response_id: string | null;
|
||||
@@ -196,6 +198,63 @@ export async function callJudge<T>(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this
|
||||
* helper (imported by many paid tests) does not pull the quarantine registry
|
||||
* into their touchfile closure. test/judge-panel.test.ts pins the two equal.
|
||||
*/
|
||||
export const JUDGE_PANEL_SAMPLES = 3;
|
||||
|
||||
/**
|
||||
* Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of
|
||||
* independent samples of the SAME prompt concurrently, inside its unchanged
|
||||
* JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
|
||||
* against the unchanged minimum; boolean fields gate on a strict majority.
|
||||
* A sample that errors (refusal, truncation, non-JSON, malformed field) fails
|
||||
* the whole panel and is never resampled. callJudge's 429 backoff happens
|
||||
* before any model output exists, so it is transport, not a verdict retry.
|
||||
*/
|
||||
export async function judgePanel<T>(sample: () => Promise<T>): Promise<T[]> {
|
||||
const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample()));
|
||||
const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []);
|
||||
if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult<T>).value);
|
||||
const first = failures[0]!;
|
||||
// A refusal is an unscored panel only when EVERY sample refused; a partial
|
||||
// refusal beside scored samples is an ordinary failed panel.
|
||||
if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) {
|
||||
throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`);
|
||||
}
|
||||
throw first.reason;
|
||||
}
|
||||
|
||||
/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */
|
||||
export function judgePanelMean<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, keys: readonly K[]): Record<K, number> {
|
||||
if (samples.length === 0) throw new Error('Judge panel has no samples');
|
||||
return Object.fromEntries(keys.map(key => {
|
||||
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
|
||||
const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value));
|
||||
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`);
|
||||
return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length];
|
||||
})) as Record<K, number>;
|
||||
}
|
||||
|
||||
/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */
|
||||
export function judgePanelMajority<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, key: K): boolean {
|
||||
if (samples.length === 0) throw new Error('Judge panel has no samples');
|
||||
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
|
||||
const bad = values.findIndex(value => typeof value !== 'boolean');
|
||||
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`);
|
||||
return values.filter(value => value === true).length * 2 > values.length;
|
||||
}
|
||||
|
||||
/** Sample reasoning lines, numbered, for the collector record. */
|
||||
export function judgePanelReasoning(samples: ReadonlyArray<unknown>): string {
|
||||
return samples.map((sample, index) => {
|
||||
const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined;
|
||||
return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`;
|
||||
}).join('\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Score documentation quality on clarity/completeness/actionability (1-5).
|
||||
*/
|
||||
|
||||
@@ -72,7 +72,12 @@ export const CASE_CI_EXCLUDE: Record<string, { reason: string; tracking: string
|
||||
* new-policy trials; exit at >= `exit.rate` over >=
|
||||
* `exit.minTrials`; at most `capFraction` of each tier's
|
||||
* blocking cases; an entry expires after `expiryWeeklyRuns`.
|
||||
* drift - one-sided Fisher exact alarm between input-identity series.
|
||||
* judge - a judge case draws `samples` independent samples of one
|
||||
* prompt concurrently; numeric dimensions gate on the panel
|
||||
* mean against the unchanged threshold, booleans on a strict
|
||||
* majority; an erroring sample fails the panel, never resampled.
|
||||
* drift - one-sided Fisher exact alarm between input-identity series
|
||||
* (Holm-controlled across the cases tested in one report).
|
||||
* infraRedispatch - a census whose every red verdict is machine-classified
|
||||
* INFRA or INCOMPLETE may be re-dispatched this many times as
|
||||
* a new run; both runs are reported.
|
||||
@@ -86,6 +91,7 @@ export const EVAL_POLICY = {
|
||||
capFraction: 0.10,
|
||||
expiryWeeklyRuns: 8,
|
||||
},
|
||||
judge: { samples: 3 },
|
||||
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
|
||||
infraRedispatch: 1,
|
||||
} as const;
|
||||
@@ -100,14 +106,18 @@ export const EVAL_POLICY = {
|
||||
* failures (a product defect is never quarantined), and unchanged case
|
||||
* touchfiles in the change that adds it. Pinned by
|
||||
* test/periodic-exclude-policy.test.ts.
|
||||
* reason - the written diagnosis
|
||||
* tracking - issue or TODOS pointer
|
||||
* owner - who removes it
|
||||
* enteredAt - ISO date the entry landed (expiry counts weekly runs from here)
|
||||
* exit - the measurable exit condition
|
||||
* reason - the written diagnosis, with the pass-rate evidence
|
||||
* failureClass - what the diagnosis found; a product defect has no class here
|
||||
* tracking - issue or TODOS pointer
|
||||
* owner - who removes it
|
||||
* enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here)
|
||||
* exit - the measurable exit condition
|
||||
* At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and
|
||||
* periodic are the blocking tiers) may be quarantined at once.
|
||||
*/
|
||||
export const CASE_QUARANTINE: Record<string, {
|
||||
reason: string;
|
||||
failureClass: 'detector' | 'harness' | 'model-latency';
|
||||
tracking: string;
|
||||
owner: string;
|
||||
enteredAt: string;
|
||||
|
||||
@@ -34,6 +34,8 @@ import {
|
||||
E2E_TIERS,
|
||||
LLM_JUDGE_TOUCHFILES,
|
||||
GLOBAL_TOUCHFILES,
|
||||
E2E_KINDS,
|
||||
BEHAVIOR_WHY,
|
||||
} from './touchfiles-data';
|
||||
|
||||
/** Repo-relative path of the pure-data file (the map-diff subject). */
|
||||
@@ -145,6 +147,9 @@ export interface TouchfileMaps {
|
||||
E2E_TIERS: Record<string, string>;
|
||||
LLM_JUDGE_TOUCHFILES: Record<string, string[]>;
|
||||
GLOBAL_TOUCHFILES: string[];
|
||||
/** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */
|
||||
E2E_KINDS?: Record<string, string>;
|
||||
BEHAVIOR_WHY?: Record<string, string>;
|
||||
}
|
||||
|
||||
export type MapDiffCause =
|
||||
@@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = {
|
||||
E2E_TIERS,
|
||||
LLM_JUDGE_TOUCHFILES,
|
||||
GLOBAL_TOUCHFILES,
|
||||
E2E_KINDS,
|
||||
BEHAVIOR_WHY,
|
||||
};
|
||||
|
||||
function isStringArray(v: unknown): v is string[] {
|
||||
@@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps {
|
||||
return isRecordOfStringArrays(o.E2E_TOUCHFILES)
|
||||
&& isRecordOfStrings(o.E2E_TIERS)
|
||||
&& isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES)
|
||||
&& isStringArray(o.GLOBAL_TOUCHFILES);
|
||||
&& isStringArray(o.GLOBAL_TOUCHFILES)
|
||||
&& (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS))
|
||||
&& (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY));
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure map-diff core (injectable for tests — no git, no filesystem).
|
||||
*
|
||||
* A key counts as CHANGED when it was added to any per-key map, its dep-list
|
||||
* array differs, or its tier value flipped. A key counts as REMOVED only when
|
||||
* array differs, or its tier, kind or behavior tolerance changed. A per-key
|
||||
* map missing on the old side (a base revision older than E2E_KINDS /
|
||||
* BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when
|
||||
* it is gone from every new per-key map; a key dropped from one map but still
|
||||
* present in another (e.g. tier entry deleted, touchfile entry kept) counts
|
||||
* as changed — conservative, because the test still exists with a different
|
||||
@@ -212,7 +223,7 @@ export function diffTouchfileMapsCore(
|
||||
oldMaps: TouchfileMaps,
|
||||
newMaps: TouchfileMaps,
|
||||
): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } {
|
||||
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const;
|
||||
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const;
|
||||
const changed = new Set<string>();
|
||||
const rawRemoved = new Set<string>();
|
||||
|
||||
@@ -290,6 +301,8 @@ export function diffTouchfileMaps(
|
||||
' E2E_TIERS: m.E2E_TIERS,',
|
||||
' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,',
|
||||
' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,',
|
||||
' E2E_KINDS: m.E2E_KINDS,',
|
||||
' BEHAVIOR_WHY: m.BEHAVIOR_WHY,',
|
||||
'}));',
|
||||
'',
|
||||
].join('\n'));
|
||||
|
||||
@@ -1597,7 +1597,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'shared-libs-unsupported-git': 'rule',
|
||||
'shared-libs-review-lifecycle': 'rule',
|
||||
'shared-libs-review-revalidation': 'rule',
|
||||
'shared-libs-opportunity-judgment': 'rule',
|
||||
'shared-libs-opportunity-judgment': 'behavior',
|
||||
'shared-libs-pr-coverage': 'rule',
|
||||
'shared-libs-plan-callers': 'rule',
|
||||
'browse-basic': 'rule',
|
||||
@@ -1634,7 +1634,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'review-sql-injection': 'rule',
|
||||
'review-enum-completeness': 'rule',
|
||||
'review-base-branch': 'rule',
|
||||
'review-design-lite': 'rule',
|
||||
'review-design-lite': 'behavior',
|
||||
'review-coverage-audit': 'rule',
|
||||
'review-dashboard-via': 'rule',
|
||||
'review-army-migration-safety': 'rule',
|
||||
@@ -1642,21 +1642,21 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'review-army-delivery-audit': 'rule',
|
||||
'review-army-quality-score': 'rule',
|
||||
'review-army-json-findings': 'rule',
|
||||
'review-army-red-team': 'rule',
|
||||
'review-army-consensus': 'rule',
|
||||
'review-army-simplification': 'rule',
|
||||
'review-army-simplification-precision': 'rule',
|
||||
'review-army-red-team': 'behavior',
|
||||
'review-army-consensus': 'behavior',
|
||||
'review-army-simplification': 'behavior',
|
||||
'review-army-simplification-precision': 'behavior',
|
||||
'office-hours-spec-review': 'rule',
|
||||
'office-hours-brain-writeback': 'rule',
|
||||
'office-hours-brain-writeback': 'behavior',
|
||||
'gbrain-roundtrip-local': 'rule',
|
||||
'sync-gbrain-read-ready': 'rule',
|
||||
'sync-gbrain-read-unknown': 'rule',
|
||||
'office-hours-forcing-energy': 'rule',
|
||||
'office-hours-builder-wildness': 'rule',
|
||||
'office-hours-forcing-energy': 'behavior',
|
||||
'office-hours-builder-wildness': 'behavior',
|
||||
'plan-ceo-review': 'rule',
|
||||
'plan-ceo-review-selective': 'rule',
|
||||
'plan-ceo-review-benefits': 'rule',
|
||||
'plan-ceo-review-expansion-energy': 'rule',
|
||||
'plan-ceo-review-expansion-energy': 'behavior',
|
||||
'plan-eng-review': 'rule',
|
||||
'plan-eng-review-artifact': 'rule',
|
||||
'plan-eng-coverage-audit': 'rule',
|
||||
@@ -1688,16 +1688,16 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'setup-gbrain-remote': 'rule',
|
||||
'setup-gbrain-bad-token': 'rule',
|
||||
'setup-gbrain-path4-local-pglite': 'rule',
|
||||
'plan-ceo-review-format-mode': 'rule',
|
||||
'plan-ceo-review-format-approach': 'rule',
|
||||
'plan-eng-review-format-coverage': 'rule',
|
||||
'plan-eng-review-format-kind': 'rule',
|
||||
'office-hours-phase4-fork': 'rule',
|
||||
'llm-judge-recommendation': 'rule',
|
||||
'plan-ceo-review-prosons-cadence': 'rule',
|
||||
'plan-review-prosons-format': 'rule',
|
||||
'plan-review-prosons-hardstop-neg': 'rule',
|
||||
'plan-review-prosons-neutral-neg': 'rule',
|
||||
'plan-ceo-review-format-mode': 'behavior',
|
||||
'plan-ceo-review-format-approach': 'behavior',
|
||||
'plan-eng-review-format-coverage': 'behavior',
|
||||
'plan-eng-review-format-kind': 'behavior',
|
||||
'office-hours-phase4-fork': 'behavior',
|
||||
'llm-judge-recommendation': 'judge',
|
||||
'plan-ceo-review-prosons-cadence': 'behavior',
|
||||
'plan-review-prosons-format': 'behavior',
|
||||
'plan-review-prosons-hardstop-neg': 'behavior',
|
||||
'plan-review-prosons-neutral-neg': 'behavior',
|
||||
'plan-tune-inspect': 'rule',
|
||||
'codex-offered-office-hours': 'rule',
|
||||
'codex-offered-ceo-review': 'rule',
|
||||
@@ -1758,7 +1758,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'design-review-detector-shim': 'rule',
|
||||
'design-review-detector-shim-dom': 'rule',
|
||||
'design-review-plugin-handoff': 'rule',
|
||||
'design-html-slop-gate': 'rule',
|
||||
'design-html-slop-gate': 'behavior',
|
||||
'diagram-triplet': 'rule',
|
||||
'diagram-authoring-quality': 'rule',
|
||||
'gstack-upgrade-happy-path': 'rule',
|
||||
@@ -1770,8 +1770,8 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'setup-deploy-workflow': 'rule',
|
||||
'autoplan-dual-voice': 'rule',
|
||||
'benchmark-providers-live': 'rule',
|
||||
'scrape-match-path': 'rule',
|
||||
'scrape-prototype-path': 'rule',
|
||||
'scrape-match-path': 'behavior',
|
||||
'scrape-prototype-path': 'behavior',
|
||||
'skillify-happy-path': 'rule',
|
||||
'skillify-provenance-refusal': 'rule',
|
||||
'skillify-approval-reject': 'rule',
|
||||
@@ -1799,30 +1799,30 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'overlay-harness-opus-4-7-literal-interpretation': 'rule',
|
||||
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule',
|
||||
'journey-negatives': 'rule',
|
||||
'review/SKILL.md workflow': 'rule',
|
||||
'setup-browser-cookies/SKILL.md workflow': 'rule',
|
||||
'browse/SKILL.md reference': 'rule',
|
||||
'setup block': 'rule',
|
||||
'qa/SKILL.md workflow': 'rule',
|
||||
'qa/SKILL.md health rubric': 'rule',
|
||||
'qa/SKILL.md anti-refusal': 'rule',
|
||||
'cross-skill greptile consistency': 'rule',
|
||||
'ship/SKILL.md workflow': 'rule',
|
||||
'document-release/SKILL.md workflow': 'rule',
|
||||
'plan-ceo-review/SKILL.md modes': 'rule',
|
||||
'plan-eng-review/SKILL.md sections': 'rule',
|
||||
'plan-design-review/SKILL.md passes': 'rule',
|
||||
'design-review/SKILL.md fix loop': 'rule',
|
||||
'design-consultation/SKILL.md research': 'rule',
|
||||
'land-and-deploy/SKILL.md workflow': 'rule',
|
||||
'canary/SKILL.md monitoring loop': 'rule',
|
||||
'benchmark/SKILL.md perf collection': 'rule',
|
||||
'setup-deploy/SKILL.md platform setup': 'rule',
|
||||
'retro/SKILL.md instructions': 'rule',
|
||||
'qa-only/SKILL.md workflow': 'rule',
|
||||
'gstack-upgrade/SKILL.md upgrade flow': 'rule',
|
||||
'sync-gbrain/SKILL.md read-only readiness': 'rule',
|
||||
'voice directive tone': 'rule',
|
||||
'review/SKILL.md workflow': 'judge',
|
||||
'setup-browser-cookies/SKILL.md workflow': 'judge',
|
||||
'browse/SKILL.md reference': 'judge',
|
||||
'setup block': 'judge',
|
||||
'qa/SKILL.md workflow': 'judge',
|
||||
'qa/SKILL.md health rubric': 'judge',
|
||||
'qa/SKILL.md anti-refusal': 'judge',
|
||||
'cross-skill greptile consistency': 'judge',
|
||||
'ship/SKILL.md workflow': 'judge',
|
||||
'document-release/SKILL.md workflow': 'judge',
|
||||
'plan-ceo-review/SKILL.md modes': 'judge',
|
||||
'plan-eng-review/SKILL.md sections': 'judge',
|
||||
'plan-design-review/SKILL.md passes': 'judge',
|
||||
'design-review/SKILL.md fix loop': 'judge',
|
||||
'design-consultation/SKILL.md research': 'judge',
|
||||
'land-and-deploy/SKILL.md workflow': 'judge',
|
||||
'canary/SKILL.md monitoring loop': 'judge',
|
||||
'benchmark/SKILL.md perf collection': 'judge',
|
||||
'setup-deploy/SKILL.md platform setup': 'judge',
|
||||
'retro/SKILL.md instructions': 'judge',
|
||||
'qa-only/SKILL.md workflow': 'judge',
|
||||
'gstack-upgrade/SKILL.md upgrade flow': 'judge',
|
||||
'sync-gbrain/SKILL.md read-only readiness': 'judge',
|
||||
'voice directive tone': 'judge',
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -1830,4 +1830,49 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
* deviation is acceptable product behavior. Keys equal the behavior ids of
|
||||
* E2E_KINDS; values are non-empty.
|
||||
*/
|
||||
export const BEHAVIOR_WHY: Record<string, string> = {};
|
||||
export const BEHAVIOR_WHY: Record<string, string> = {
|
||||
'shared-libs-opportunity-judgment':
|
||||
"Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.",
|
||||
'review-design-lite':
|
||||
"How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.",
|
||||
'review-army-red-team':
|
||||
"Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.",
|
||||
'review-army-consensus':
|
||||
"Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.",
|
||||
'review-army-simplification':
|
||||
"Flagging the planted unnecessary structure is an advisory-lens quality judgment.",
|
||||
'review-army-simplification-precision':
|
||||
"Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.",
|
||||
'office-hours-forcing-energy':
|
||||
"The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.",
|
||||
'office-hours-builder-wildness':
|
||||
"Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.",
|
||||
'office-hours-brain-writeback':
|
||||
"The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.",
|
||||
'office-hours-phase4-fork':
|
||||
"Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.",
|
||||
'plan-ceo-review-expansion-energy':
|
||||
"Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.",
|
||||
'plan-ceo-review-format-mode':
|
||||
"Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.",
|
||||
'plan-ceo-review-format-approach':
|
||||
"Approach-menu Completeness wording is live formatting of one AskUserQuestion.",
|
||||
'plan-eng-review-format-coverage':
|
||||
"Coverage-issue Completeness wording is live formatting of one AskUserQuestion.",
|
||||
'plan-eng-review-format-kind':
|
||||
"Kind-note wording is live formatting of one AskUserQuestion.",
|
||||
'plan-ceo-review-prosons-cadence':
|
||||
"Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.",
|
||||
'plan-review-prosons-format':
|
||||
"The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.",
|
||||
'plan-review-prosons-hardstop-neg':
|
||||
"Not using the hard-stop escape on an ordinary decision is live formatting of one question.",
|
||||
'plan-review-prosons-neutral-neg':
|
||||
"Avoiding neutral posture and naming a because-reason is live formatting of one question.",
|
||||
'design-html-slop-gate':
|
||||
"How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.",
|
||||
'scrape-match-path':
|
||||
"The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.",
|
||||
'scrape-prototype-path':
|
||||
"The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.",
|
||||
};
|
||||
@@ -4,7 +4,7 @@ import * as path from 'node:path';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
|
||||
import { JUDGE_MS } from './eval-budgets';
|
||||
import type { JudgeScore } from './llm-judge';
|
||||
import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge';
|
||||
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input';
|
||||
import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache,
|
||||
type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache';
|
||||
@@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres
|
||||
|| typeof value.reasoning !== 'string'
|
||||
|| (structuredResponse && (!value.reasoning.trim()
|
||||
|| value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false;
|
||||
return (['clarity', 'completeness', 'actionability'] as const).every(key =>
|
||||
return JUDGE_SCORE_DIMENSIONS.every(key =>
|
||||
typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5);
|
||||
}
|
||||
|
||||
const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 };
|
||||
|
||||
/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */
|
||||
export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array<JudgeScore & EvalCacheValue> } {
|
||||
if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples'
|
||||
|| !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES
|
||||
|| !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false;
|
||||
const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS);
|
||||
return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]);
|
||||
}
|
||||
|
||||
export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null;
|
||||
lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null;
|
||||
/** The attempt guard is rechecked after synchronous input/provenance reads. */
|
||||
publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined;
|
||||
publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined;
|
||||
} {
|
||||
const env = opts.env ?? process.env;
|
||||
const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined };
|
||||
const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined };
|
||||
const pr = Number(env.EVALS_CACHE_PR);
|
||||
// Runtime ID is the immutable CI image manifest, not a mutable image tag.
|
||||
// Nonstandard Node/Bun preload code or custom model endpoints need a separate
|
||||
@@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
|
||||
prompts: { [opts.testName]: prompt },
|
||||
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
|
||||
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1,
|
||||
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0,
|
||||
panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' },
|
||||
...(opts.stream ? { stream: true } : {}),
|
||||
...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
|
||||
response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) },
|
||||
@@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
return {
|
||||
lookup() {
|
||||
const result = lookupEvalInputCache({ ...common, identity: before,
|
||||
validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) });
|
||||
validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) });
|
||||
return result.status === 'reused'
|
||||
? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null;
|
||||
? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null;
|
||||
},
|
||||
publish(scores, isActive = () => true) {
|
||||
publish(samples, isActive = () => true) {
|
||||
// Caller reaches here ONLY after its actual assertions passed. A later
|
||||
// failed case in the file does not erase this independently completed case.
|
||||
if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
|
||||
const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) };
|
||||
if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
|
||||
const after = currentIdentity();
|
||||
const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID;
|
||||
if (!after || !runId || !isActive()) return;
|
||||
@@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
cancelled: false, skipped: 0, failed: 0, passed: 1,
|
||||
cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }],
|
||||
source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() },
|
||||
result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning },
|
||||
result: panel,
|
||||
} });
|
||||
// A slow synchronous write can consume the recording allowance. The
|
||||
// caller withdraws this new receipt if its final deadline check fails.
|
||||
|
||||
@@ -9,10 +9,11 @@ import { describe, expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
|
||||
import { CASE_CI_EXCLUDE, PERIODIC_CI_EXCLUDE } from './helpers/periodic-exclude-data';
|
||||
import { CASE_CI_EXCLUDE, CASE_QUARANTINE, EVAL_POLICY, PERIODIC_CI_EXCLUDE } from './helpers/periodic-exclude-data';
|
||||
import { E2E_TOUCHFILES } from './helpers/touchfiles';
|
||||
import { quarantinePolicyProblems } from '../scripts/eval-flake-rank';
|
||||
import { isPaidTestFile } from './helpers/paid-test-set';
|
||||
import { buildRunManifest, CASE_SHARDED_FILES, expandCaseShards, partitionCaseExclusions, selectPaidTestFiles, shardCaseId, shardFile } from '../scripts/test-paid-shards';
|
||||
import { buildRunManifest, CASE_SHARDED_FILES, expandCaseShards, fileCaseRegistration, partitionCaseExclusions, selectPaidTestFiles, shardCaseId, shardFile } from '../scripts/test-paid-shards';
|
||||
|
||||
const ROOT = path.resolve(__dirname, '..');
|
||||
|
||||
@@ -72,3 +73,30 @@ describe('periodic exclude policy', () => {
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('eval verdict policy (pre-registered)', () => {
|
||||
test('EVAL_POLICY carries exactly the approved constants; a change needs re-approval and a version bump', () => {
|
||||
expect(EVAL_POLICY).toEqual({
|
||||
version: 1,
|
||||
panel: { n: 3, k: 2 },
|
||||
quarantine: { entry: { rate: 0.95, minTrials: 10 }, exit: { rate: 0.97, minTrials: 10 }, capFraction: 0.10, expiryWeeklyRuns: 8 },
|
||||
judge: { samples: 3 },
|
||||
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
|
||||
infraRedispatch: 1,
|
||||
});
|
||||
});
|
||||
|
||||
test('every CASE_QUARANTINE entry is a diagnosed, dated, non-product blocking case within the tier cap', () => {
|
||||
expect(quarantinePolicyProblems(CASE_QUARANTINE).map(problem => problem.message)).toEqual([]);
|
||||
});
|
||||
|
||||
test('a quarantined case runs as isolated trial shards: its files register it literally', () => {
|
||||
for (const id of Object.keys(CASE_QUARANTINE)) {
|
||||
const files = (E2E_TOUCHFILES[id] ?? []).filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file));
|
||||
expect(files.length, `${id}: no paid file registers it`).toBeGreaterThan(0);
|
||||
for (const file of files) {
|
||||
expect(fileCaseRegistration(file, fs.readFileSync(path.join(ROOT, file), 'utf8')).known, `${id}: ${file} registration must be statically known`).toBe(true);
|
||||
}
|
||||
}
|
||||
});
|
||||
});
|
||||
+55
-46
@@ -14,7 +14,7 @@ import { afterAll, expect } from 'bun:test';
|
||||
import { JUDGE_MS } from './helpers/eval-budgets';
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
|
||||
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMean, judgePanelMajority, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS } from './helpers/llm-judge';
|
||||
import { ENG_REVIEW_EXCERPT } from './helpers/workflow-excerpt';
|
||||
import type { JudgeScore } from './helpers/llm-judge';
|
||||
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA, type WorkflowJudgeInput } from './helpers/workflow-judge-input';
|
||||
@@ -100,8 +100,9 @@ describeIfSelected('LLM-as-judge quality evals', [
|
||||
// rewrites the pin).
|
||||
const section = sliceBrowseSection('## Snapshot Flags');
|
||||
|
||||
const scores = await judge('browse skill reference (flags + commands)', section);
|
||||
console.log('Browse SKILL.md scores:', JSON.stringify(scores, null, 2));
|
||||
const samples = await judgePanel(() => judge('browse skill reference (flags + commands)', section));
|
||||
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
|
||||
console.log('Browse SKILL.md panel:', JSON.stringify({ mean: scores, samples }, null, 2));
|
||||
|
||||
const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json');
|
||||
const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8'));
|
||||
@@ -120,9 +121,9 @@ describeIfSelected('LLM-as-judge quality evals', [
|
||||
tier: 'llm-judge',
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4 && regressions.length === 0,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
cost_usd: 0.02 * samples.length,
|
||||
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: regressions.length ? `${scores.reasoning} | ${regressions.join('; ')}` : scores.reasoning,
|
||||
judge_reasoning: regressions.length ? `${judgePanelReasoning(samples)} | ${regressions.join('; ')}` : judgePanelReasoning(samples),
|
||||
});
|
||||
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
@@ -144,8 +145,9 @@ describeIfSelected('LLM-as-judge quality evals', [
|
||||
if (setupStart < 0 || setupEnd < 0) throw new Error('browse/SKILL.md: setup block not found — regenerate with: bun run gen:skill-docs');
|
||||
const section = content.slice(setupStart, setupEnd);
|
||||
|
||||
const scores = await judge('setup/binary discovery instructions', section);
|
||||
console.log('Setup block scores:', JSON.stringify(scores, null, 2));
|
||||
const samples = await judgePanel(() => judge('setup/binary discovery instructions', section));
|
||||
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
|
||||
console.log('Setup block panel:', JSON.stringify({ mean: scores, samples }, null, 2));
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'setup block',
|
||||
@@ -153,9 +155,9 @@ describeIfSelected('LLM-as-judge quality evals', [
|
||||
tier: 'llm-judge',
|
||||
passed: scores.actionability >= 3 && scores.clarity >= 3,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
cost_usd: 0.02 * samples.length,
|
||||
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: scores.reasoning,
|
||||
judge_reasoning: judgePanelReasoning(samples),
|
||||
});
|
||||
|
||||
// Setup block is intentionally minimal (binary discovery only).
|
||||
@@ -203,7 +205,7 @@ describeIfSelected('QA skill quality evals', ['qa/SKILL.md workflow', 'qa/SKILL.
|
||||
startMarker: '# /qa: Test', endMarker: null,
|
||||
references: ['qa/templates/functional-report-template.md'] }).text;
|
||||
|
||||
const scores = await callJudge<JudgeScore>(`You are evaluating the quality of a QA testing workflow document for an AI coding agent.
|
||||
const samples = await judgePanel(() => callJudge<JudgeScore>(`You are evaluating the quality of a QA testing workflow document for an AI coding agent.
|
||||
|
||||
The agent reads this source-file bundle to select browser, native functional or mixed
|
||||
surfaces, explore with bounded probes, reproduce and diagnose defects, add a regression
|
||||
@@ -222,8 +224,9 @@ Respond with ONLY valid JSON:
|
||||
|
||||
Here is the QA workflow to evaluate:
|
||||
|
||||
${section}`);
|
||||
console.log('QA workflow scores:', JSON.stringify(scores, null, 2));
|
||||
${section}`));
|
||||
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
|
||||
console.log('QA workflow panel:', JSON.stringify({ mean: scores, samples }, null, 2));
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'qa/SKILL.md workflow',
|
||||
@@ -231,9 +234,9 @@ ${section}`);
|
||||
tier: 'llm-judge',
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
cost_usd: 0.02 * samples.length,
|
||||
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: scores.reasoning,
|
||||
judge_reasoning: judgePanelReasoning(samples),
|
||||
});
|
||||
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
@@ -247,7 +250,7 @@ ${section}`);
|
||||
const t0 = Date.now();
|
||||
const section = sliceQaPatterns('## Health Score Rubric');
|
||||
|
||||
const scores = await callJudge<JudgeScore>(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score.
|
||||
const samples = await judgePanel(() => callJudge<JudgeScore>(`You are evaluating a health score rubric that an AI agent must follow to compute a numeric QA score.
|
||||
|
||||
The agent uses this rubric after QA testing a website. It needs to:
|
||||
1. Understand each scoring category and what counts as a deduction
|
||||
@@ -264,8 +267,9 @@ Respond with ONLY valid JSON:
|
||||
|
||||
Here is the rubric to evaluate:
|
||||
|
||||
${section}`);
|
||||
console.log('QA health rubric scores:', JSON.stringify(scores, null, 2));
|
||||
${section}`));
|
||||
const scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
|
||||
console.log('QA health rubric panel:', JSON.stringify({ mean: scores, samples }, null, 2));
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'qa/SKILL.md health rubric',
|
||||
@@ -273,9 +277,9 @@ ${section}`);
|
||||
tier: 'llm-judge',
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
cost_usd: 0.02 * samples.length,
|
||||
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: scores.reasoning,
|
||||
judge_reasoning: judgePanelReasoning(samples),
|
||||
});
|
||||
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
@@ -294,7 +298,7 @@ ${section}`);
|
||||
const diffAwareSection = sliceQaPatterns('### Diff-aware', '### Full');
|
||||
const rulesSection = sliceQaPatterns('## Important Rules');
|
||||
|
||||
const result = await callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario.
|
||||
const samples = await judgePanel(() => callJudge<{ would_browse: boolean; fallback_behavior: string; confidence: number; reasoning: string }>(`You are evaluating whether a QA testing skill document would cause an AI agent to USE THE BROWSER or REFUSE to use the browser in a specific scenario.
|
||||
|
||||
SCENARIO:
|
||||
A user runs /qa (a browser-based QA testing skill). The branch diff shows ONLY prompt template files and config file changes — no routes, views, controllers, components, or CSS were changed. The changes are "purely backend" with no obvious UI surface.
|
||||
@@ -318,9 +322,10 @@ Respond with ONLY valid JSON:
|
||||
Rules:
|
||||
- would_browse should be true if the document instructs the agent to always use the browser regardless of diff content
|
||||
- would_browse should be false if the document allows the agent to skip browser testing for non-UI changes
|
||||
- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`);
|
||||
- confidence: 5 = document is unambiguous, 1 = document is unclear or contradictory`));
|
||||
const result = { would_browse: judgePanelMajority(samples, 'would_browse'), ...judgePanelMean(samples, ['confidence'] as const) };
|
||||
|
||||
console.log('QA anti-refusal result:', JSON.stringify(result, null, 2));
|
||||
console.log('QA anti-refusal panel:', JSON.stringify({ result, samples }, null, 2));
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'qa/SKILL.md anti-refusal',
|
||||
@@ -328,9 +333,9 @@ Rules:
|
||||
tier: 'llm-judge',
|
||||
passed: result.would_browse === true && result.confidence >= 4,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
cost_usd: 0.02 * samples.length,
|
||||
judge_scores: { would_browse: result.would_browse ? 1 : 0, confidence: result.confidence },
|
||||
judge_reasoning: result.reasoning,
|
||||
judge_reasoning: judgePanelReasoning(samples),
|
||||
});
|
||||
|
||||
expect(result.would_browse).toBe(true);
|
||||
@@ -362,7 +367,7 @@ describeIfSelected('Cross-skill consistency evals', ['cross-skill greptile consi
|
||||
extractGrepLines(retroContent, 'retro/SKILL.md'),
|
||||
].join('\n\n');
|
||||
|
||||
const result = await callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently.
|
||||
const samples = await judgePanel(() => callJudge<{ consistent: boolean; issues: string[]; score: number; reasoning: string }>(`You are evaluating whether multiple skill configuration files implement the same data architecture consistently.
|
||||
|
||||
INTENDED ARCHITECTURE:
|
||||
- greptile-history has TWO paths: per-project (~/.gstack/projects/{slug}/greptile-history.md) and global (~/.gstack/greptile-history.md)
|
||||
@@ -383,9 +388,10 @@ Evaluate consistency. Respond with ONLY valid JSON:
|
||||
"reasoning": "brief explanation"
|
||||
}
|
||||
|
||||
score (1-5): 5 = perfectly consistent, 1 = contradictory`);
|
||||
score (1-5): 5 = perfectly consistent, 1 = contradictory`));
|
||||
const result = { consistent: judgePanelMajority(samples, 'consistent'), ...judgePanelMean(samples, ['score'] as const) };
|
||||
|
||||
console.log('Cross-skill consistency:', JSON.stringify(result, null, 2));
|
||||
console.log('Cross-skill consistency panel:', JSON.stringify({ result, samples }, null, 2));
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'cross-skill greptile consistency',
|
||||
@@ -393,9 +399,9 @@ score (1-5): 5 = perfectly consistent, 1 = contradictory`);
|
||||
tier: 'llm-judge',
|
||||
passed: result.consistent && result.score >= 4,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
cost_usd: 0.02 * samples.length,
|
||||
judge_scores: { consistency_score: result.score },
|
||||
judge_reasoning: result.reasoning,
|
||||
judge_reasoning: judgePanelReasoning(samples),
|
||||
});
|
||||
|
||||
expect(result.consistent).toBe(true);
|
||||
@@ -439,7 +445,8 @@ async function runWorkflowJudge(opts: {
|
||||
const workDeadline = started + JUDGE_MS;
|
||||
let stage: 'input' | 'judge' | 'validation' | 'recording' = 'input';
|
||||
let finalized = false;
|
||||
let scores: JudgeScore | undefined;
|
||||
let samples: JudgeScore[] | undefined;
|
||||
let scores: Record<typeof JUDGE_SCORE_DIMENSIONS[number], number> | undefined;
|
||||
let manualReview: ManualJudgeReview | undefined;
|
||||
let customInputMetadata: { prompt: string; model: string } | undefined;
|
||||
let reused: ReturnType<ReturnType<typeof prepareWorkflowJudgeCache>['lookup']> = null;
|
||||
@@ -458,19 +465,19 @@ async function runWorkflowJudge(opts: {
|
||||
evalCollector?.addTest({
|
||||
name: opts.testName, suite: opts.suite, tier: 'llm-judge', passed, attempt,
|
||||
duration_ms: Math.max(0, performance.now() - started),
|
||||
cost_usd: reused || !scores ? 0 : 0.02,
|
||||
cost_usd: reused || !samples ? 0 : 0.02 * samples.length,
|
||||
execution: reused ? 'reused' : 'executed',
|
||||
...customInputMetadata,
|
||||
...(manualReview ? { manual_review: manualReview } : {}),
|
||||
...(reused ? { reused_from: { input_key: reused.reuse.key, run_id: reused.reuse.source.runId,
|
||||
revision: reused.reuse.source.revision, completed_at: new Date(reused.reuse.source.completedAt).toISOString() } } : {}),
|
||||
...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: scores.reasoning } : {}),
|
||||
...(scores ? { judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability } } : {}),
|
||||
...(samples ? { judge_reasoning: judgePanelReasoning(samples) } : {}),
|
||||
...(passed ? {} : { exit_reason: error instanceof JudgeRefusalError ? 'provider_refusal'
|
||||
: error instanceof Error && error.name === 'WorkflowJudgeDeadline' ? 'timeout'
|
||||
: error instanceof Error && error.name === 'WorkflowJudgeSuperseded' ? 'cancelled'
|
||||
: stage === 'validation' ? 'validation_failed' : 'harness_error',
|
||||
error: `${error instanceof Error ? error.message : String(error)}${scores ? '' : error instanceof JudgeRefusalError
|
||||
error: `${error instanceof Error ? error.message : String(error)}${samples ? '' : error instanceof JudgeRefusalError
|
||||
? '\nNo automated score; provider refusal usage retained when manually accepted; cost unavailable.'
|
||||
: '\nNo completed model response; cost and usage unavailable.'}` }),
|
||||
});
|
||||
@@ -508,11 +515,11 @@ async function runWorkflowJudge(opts: {
|
||||
checkActive();
|
||||
stage = 'judge';
|
||||
const maxTokens = opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS;
|
||||
let result: JudgeScore;
|
||||
let result: JudgeScore[];
|
||||
try {
|
||||
result = reused?.scores ?? await callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,
|
||||
result = reused?.samples ?? await judgePanel(() => callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,
|
||||
...(opts.stream ? { stream: true } : {}),
|
||||
...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) });
|
||||
...(opts.structuredResponse ? { jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } : {}) }));
|
||||
} catch (error) {
|
||||
checkActive();
|
||||
if (error instanceof JudgeRefusalError && customInputMetadata) {
|
||||
@@ -529,20 +536,21 @@ async function runWorkflowJudge(opts: {
|
||||
throw error;
|
||||
}
|
||||
checkActive();
|
||||
scores = result;
|
||||
samples = result;
|
||||
console.log(`[workflow-judge] ${opts.testName}: ${reused ? `reused ${reused.reuse.source.runId} @ ${reused.reuse.source.revision} (${new Date(reused.reuse.source.completedAt).toISOString()})` : 'executed'}`);
|
||||
console.log(`${opts.testName} scores:`, JSON.stringify(scores, null, 2));
|
||||
stage = 'validation';
|
||||
if (opts.structuredResponse && !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true)) {
|
||||
if (opts.structuredResponse && !samples.every(sample => validWorkflowJudgeScore(sample as unknown as EvalCacheValue, { clarity: 1, completeness: 1, actionability: 1 }, true))) {
|
||||
throw new Error('Structured workflow judge violated the response schema');
|
||||
}
|
||||
scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);
|
||||
console.log(`${opts.testName} panel:`, JSON.stringify({ mean: scores, samples }, null, 2));
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(thresholds.clarity);
|
||||
expect(scores.completeness).toBeGreaterThanOrEqual(thresholds.completeness);
|
||||
expect(scores.actionability).toBeGreaterThanOrEqual(thresholds.actionability);
|
||||
checkActive();
|
||||
stage = 'recording';
|
||||
arm();
|
||||
const discardReceipt = reused ? undefined : cache.publish(scores, active);
|
||||
const discardReceipt = reused ? undefined : cache.publish(samples, active);
|
||||
try { checkActive(); finish(true); }
|
||||
catch (error) { discardReceipt?.(); throw error; }
|
||||
};
|
||||
@@ -792,7 +800,7 @@ describeIfSelected('Voice directive eval', ['voice directive tone'], () => {
|
||||
const voiceEnd = content.indexOf('\n## ', voiceStart + 1);
|
||||
const voiceSection = content.slice(voiceStart, voiceEnd > 0 ? voiceEnd : voiceStart + 3000);
|
||||
|
||||
const result = await callJudge<{
|
||||
const samples = await judgePanel(() => callJudge<{
|
||||
directness: number;
|
||||
concreteness: number;
|
||||
avoids_corporate: number;
|
||||
@@ -812,9 +820,10 @@ Return JSON only:
|
||||
{"directness": N, "concreteness": N, "avoids_corporate": N, "avoids_ai_vocabulary": N, "connects_user_outcomes": N, "reasoning": "..."}
|
||||
|
||||
THE VOICE DIRECTIVE:
|
||||
${voiceSection}`);
|
||||
${voiceSection}`));
|
||||
const result = judgePanelMean(samples, ['directness', 'concreteness', 'avoids_corporate', 'avoids_ai_vocabulary', 'connects_user_outcomes'] as const);
|
||||
|
||||
console.log('Voice directive scores:', JSON.stringify(result, null, 2));
|
||||
console.log('Voice directive panel:', JSON.stringify({ mean: result, samples }, null, 2));
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'voice directive tone',
|
||||
@@ -823,7 +832,7 @@ ${voiceSection}`);
|
||||
passed: result.directness >= 4 && result.concreteness >= 4 && result.avoids_corporate >= 4
|
||||
&& result.avoids_ai_vocabulary >= 4 && result.connects_user_outcomes >= 4,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
cost_usd: 0.02 * samples.length,
|
||||
judge_scores: {
|
||||
directness: result.directness,
|
||||
concreteness: result.concreteness,
|
||||
@@ -831,7 +840,7 @@ ${voiceSection}`);
|
||||
avoids_ai_vocabulary: result.avoids_ai_vocabulary,
|
||||
connects_user_outcomes: result.connects_user_outcomes,
|
||||
},
|
||||
judge_reasoning: result.reasoning,
|
||||
judge_reasoning: judgePanelReasoning(samples),
|
||||
});
|
||||
|
||||
expect(result.directness).toBeGreaterThanOrEqual(4);
|
||||
|
||||
@@ -1,18 +1,22 @@
|
||||
import { afterEach, expect, spyOn, test } from 'bun:test';
|
||||
import { afterEach, describe, expect, spyOn, test } from 'bun:test';
|
||||
import { Messages } from '@anthropic-ai/sdk/resources/messages';
|
||||
import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
|
||||
import { callJudge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS, judgePanel, judgePanelMajority, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS, JUDGE_PANEL_SAMPLES } from './helpers/llm-judge';
|
||||
import { EVAL_POLICY } from './helpers/periodic-exclude-data';
|
||||
import { getCookieWorkflowManualReview } from './helpers/cookie-workflow-manual-review';
|
||||
import { resolveEvalModel } from '../lib/eval-model';
|
||||
import * as fs from 'node:fs';
|
||||
import * as os from 'node:os';
|
||||
import * as path from 'node:path';
|
||||
import { execFileSync } from 'node:child_process';
|
||||
import { prepareWorkflowJudgeCache, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
|
||||
import { prepareWorkflowJudgeCache, validWorkflowJudgePanel, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
|
||||
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA } from './helpers/workflow-judge-input';
|
||||
|
||||
const roots: string[] = [];
|
||||
afterEach(() => { for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); });
|
||||
const scores = { clarity: 4, completeness: 5, actionability: 4, reasoning: 'Concrete steps' };
|
||||
const SAMPLES = JUDGE_PANEL_SAMPLES;
|
||||
const panelOf = (sample: typeof scores) => Array.from({ length: SAMPLES }, () => sample);
|
||||
const panel = panelOf(scores);
|
||||
function fixture() {
|
||||
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-judge-cache-')); roots.push(root);
|
||||
const files = {
|
||||
@@ -53,9 +57,9 @@ function fixture() {
|
||||
}
|
||||
|
||||
test('the audited adapter reuses only the exact completed score and original provenance', () => {
|
||||
const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(scores);
|
||||
const f = fixture(); const first = f.cache(); expect(first.lookup()).toBeNull(); first.publish(panel);
|
||||
expect(f.entries()).toHaveLength(1);
|
||||
const reused = f.cache().lookup(); expect(reused?.scores).toEqual(scores);
|
||||
const reused = f.cache().lookup(); expect(reused?.samples).toEqual(panel);
|
||||
expect(reused?.reuse.source.runId).toBe('free-cache-test');
|
||||
expect(reused?.reuse.source.revision).toMatch(/^[a-f0-9]{40}$/);
|
||||
expect(reused?.reuse.source.completedAt).toBeLessThanOrEqual(Date.now());
|
||||
@@ -73,11 +77,11 @@ test('the dependency closure includes actual installed SDK bytes and local trans
|
||||
});
|
||||
|
||||
test('release-label changes preserve reuse; other package semantics invalidate it', () => {
|
||||
const f = fixture(); f.cache().publish(scores);
|
||||
const f = fixture(); f.cache().publish(panel);
|
||||
const file = path.join(f.root, 'package.json');
|
||||
const original = JSON.parse(fs.readFileSync(file, 'utf8'));
|
||||
fs.writeFileSync(file, JSON.stringify({ ...original, version: '2.0.0' }, null, 2));
|
||||
expect(f.cache().lookup()?.scores).toEqual(scores);
|
||||
expect(f.cache().lookup()?.samples).toEqual(panel);
|
||||
for (const change of [{ scripts: { 'test:gate': 'changed command' } }, { dependencies: { 'some-sdk': '2.0.0' } }]) {
|
||||
fs.writeFileSync(file, JSON.stringify({ ...original, ...change, version: '2.0.0' }));
|
||||
expect(f.cache().lookup()).toBeNull();
|
||||
@@ -89,7 +93,7 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in
|
||||
'scripts/test-paid-shards.ts', 'scripts/test-strict-output.ts', 'scripts/eval-select.ts',
|
||||
'scripts/test-pr-profile.ts', '.github/workflows/evals.yml']) {
|
||||
test(`changes in ${file} require new evaluation`, () => {
|
||||
const f = fixture(); f.cache().publish(scores); const target = path.join(f.root, file);
|
||||
const f = fixture(); f.cache().publish(panel); const target = path.join(f.root, file);
|
||||
fs.appendFileSync(target, file.endsWith('.json') ? ' ' : '\n// changed');
|
||||
f.refreshPrompt(); expect(f.cache().lookup()).toBeNull();
|
||||
});
|
||||
@@ -98,8 +102,8 @@ for (const file of ['test/helpers/nested.ts', 'node_modules/@anthropic-ai/sdk/in
|
||||
test('changed sources during an attempt and mismatched actual prompt cannot publish', () => {
|
||||
const f = fixture(); const before = f.cache();
|
||||
fs.appendFileSync(path.join(f.root, 'example/sections/review.md'), 'new finding');
|
||||
before.publish(scores); expect(f.entries()).toHaveLength(0);
|
||||
f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(scores);
|
||||
before.publish(panel); expect(f.entries()).toHaveLength(0);
|
||||
f.refreshPrompt(); f.opts.prompt += ' hidden new request'; f.cache().publish(panel);
|
||||
expect(f.entries()).toHaveLength(0);
|
||||
});
|
||||
|
||||
@@ -108,52 +112,52 @@ for (const [key, value] of Object.entries({ EVALS_FRESH: '1', EVALS_TIER: 'perio
|
||||
EVALS_CACHE_REPOSITORY: '', NODE_OPTIONS: '--require=unknown', BUN_OPTIONS: '--preload=unknown',
|
||||
ANTHROPIC_BASE_URL: 'https://custom-provider.example.test' })) {
|
||||
test(`${key}=${value} is fresh or ineligible`, () => {
|
||||
const f = fixture(); f.cache().publish(scores);
|
||||
const f = fixture(); f.cache().publish(panel);
|
||||
f.opts.env = { ...f.env, [key]: value }; const cache = f.cache();
|
||||
expect(cache.lookup()).toBeNull(); cache.publish(scores); expect(f.entries()).toHaveLength(1);
|
||||
expect(cache.lookup()).toBeNull(); cache.publish(panel); expect(f.entries()).toHaveLength(1);
|
||||
});
|
||||
}
|
||||
|
||||
test('runtime/model/threshold changes miss, and retries never reuse or publish', () => {
|
||||
const f = fixture(); f.cache().publish(scores);
|
||||
const f = fixture(); f.cache().publish(panel);
|
||||
for (const overrides of [{ GSTACK_EVAL_MODEL_JUDGE: 'different-model' }, { EVALS_CACHE_RUNTIME_ID: 'c'.repeat(64) }]) {
|
||||
f.opts.env = { ...f.env, ...overrides }; expect(f.cache().lookup()).toBeNull();
|
||||
}
|
||||
f.opts.env = f.env; f.opts.thresholds.clarity = 5; expect(f.cache().lookup()).toBeNull();
|
||||
f.opts.thresholds.clarity = 4; f.opts.attempt = 2; const retry = f.cache();
|
||||
expect(retry.lookup()).toBeNull(); retry.publish(scores); expect(f.entries()).toHaveLength(1);
|
||||
expect(retry.lookup()).toBeNull(); retry.publish(panel); expect(f.entries()).toHaveLength(1);
|
||||
});
|
||||
|
||||
test('frontier reader calibration cannot reuse a score from the unspecified-reader rubric', () => {
|
||||
const f = fixture(); f.cache().publish(scores);
|
||||
const f = fixture(); f.cache().publish(panel);
|
||||
const original = f.opts.prompt;
|
||||
f.opts.agentCapability = 'frontier'; f.refreshPrompt();
|
||||
expect(f.opts.prompt).not.toBe(original);
|
||||
expect(f.cache().lookup()).toBeNull();
|
||||
f.cache().publish(scores);
|
||||
f.cache().publish(panel);
|
||||
expect(f.entries()).toHaveLength(2);
|
||||
expect(f.cache().lookup()?.scores).toEqual(scores);
|
||||
expect(f.cache().lookup()?.samples).toEqual(panel);
|
||||
delete f.opts.agentCapability; f.refreshPrompt();
|
||||
expect(f.opts.prompt).toBe(original);
|
||||
expect(f.cache().lookup()?.scores).toEqual(scores);
|
||||
expect(f.cache().lookup()?.samples).toEqual(panel);
|
||||
});
|
||||
|
||||
test('a pinned workflow judge model overrides the global model and changes the cache identity', () => {
|
||||
const f = fixture();
|
||||
f.opts.model = 'claude-sonnet-4-6';
|
||||
f.cache().publish(scores);
|
||||
f.cache().publish(panel);
|
||||
expect(f.entries()).toHaveLength(1);
|
||||
f.opts.env = { ...f.env, GSTACK_EVAL_MODEL_JUDGE: 'different-global-model' };
|
||||
expect(f.cache().lookup()?.scores).toEqual(scores);
|
||||
expect(f.cache().lookup()?.samples).toEqual(panel);
|
||||
f.opts.model = 'claude-opus-4-7';
|
||||
expect(f.cache().lookup()).toBeNull();
|
||||
});
|
||||
|
||||
test('failed assertions, missing provenance, and missing imported dependencies cannot supply a receipt', () => {
|
||||
const f = fixture(); f.cache().publish({ ...scores, clarity: 3 }); expect(f.entries()).toHaveLength(0);
|
||||
f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(scores); expect(f.entries()).toHaveLength(0);
|
||||
const f = fixture(); f.cache().publish(panelOf({ ...scores, clarity: 3 })); expect(f.entries()).toHaveLength(0);
|
||||
f.opts.env = { ...f.env, EVALS_RUN_ID: '' }; f.cache().publish(panel); expect(f.entries()).toHaveLength(0);
|
||||
f.opts.env = f.env; fs.unlinkSync(path.join(f.root, 'test/helpers/nested.ts'));
|
||||
f.cache().publish(scores); expect(f.entries()).toHaveLength(0);
|
||||
f.cache().publish(panel); expect(f.entries()).toHaveLength(0);
|
||||
});
|
||||
|
||||
test('cached payload schema remains small and cannot carry operational fields', () => {
|
||||
@@ -167,8 +171,9 @@ test('workflow registration preserves model work and reserves only terminal-reco
|
||||
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
|
||||
const body = source.split('async function runWorkflowJudge')[1]!.split('// Block 1:')[0]!;
|
||||
const stages = ['workflowJudgeAttempts.set', 'readWorkflowJudgeInput(', 'cache.lookup()',
|
||||
'callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
|
||||
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(scores, active)']
|
||||
'judgePanel(() => callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
|
||||
'scores = judgePanelMean(samples, JUDGE_SCORE_DIMENSIONS);',
|
||||
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(samples, active)']
|
||||
.map(stage => body.indexOf(stage));
|
||||
expect(stages.every(position => position >= 0)).toBe(true);
|
||||
expect(stages).toEqual([...stages].sort((a, b) => a - b));
|
||||
@@ -203,6 +208,7 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
|
||||
'evalCollector', 'expect', 'console', 'performance', 'JUDGE_MS', 'WORKFLOW_JUDGE_RECORD_MS',
|
||||
'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'resolveEvalModel',
|
||||
'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore',
|
||||
'judgePanel', 'judgePanelMean', 'judgePanelReasoning', 'JUDGE_SCORE_DIMENSIONS',
|
||||
`${javascript}\nreturn runWorkflowJudge;`)(
|
||||
f.root, overrides.read ?? readWorkflowJudgeInput, buildWorkflowJudgePrompt,
|
||||
(options: WorkflowCacheOptions) => (overrides.prepare ?? prepareWorkflowJudgeCache)({ ...options, env: f.env }),
|
||||
@@ -213,7 +219,8 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
|
||||
overrides.clock ? { now: overrides.clock } : performance, overrides.budget ?? 120_000, overrides.allowance ?? 5_000,
|
||||
overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout,
|
||||
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel,
|
||||
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore);
|
||||
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore,
|
||||
judgePanel, judgePanelMean, judgePanelReasoning, JUDGE_SCORE_DIMENSIONS);
|
||||
return { run, records, signals, prompts, attempts, options: { ...f.opts, suite: 'Cache regression' } };
|
||||
}
|
||||
|
||||
@@ -223,7 +230,7 @@ test('the actual workflow callback preserves the pinned model and frontier rubri
|
||||
const actual = actualCallback(f, { judge: async (_prompt, model) => { models.push(model); return scores; } });
|
||||
await actual.run({ ...actual.options, model: 'claude-sonnet-4-6', agentCapability: 'frontier',
|
||||
readInput: () => readWorkflowJudgeInput(f.opts) });
|
||||
expect(models).toEqual(['claude-sonnet-4-6']);
|
||||
expect(models).toEqual(Array(SAMPLES).fill('claude-sonnet-4-6'));
|
||||
expect(actual.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
|
||||
expect(actual.records[0]).toMatchObject({ passed: true, model: 'claude-sonnet-4-6', prompt: actual.prompts[0] });
|
||||
});
|
||||
@@ -239,7 +246,7 @@ test.each(['ship', 'review'])('the registered %s callback sends the frontier rub
|
||||
endMarker: f.opts.endMarker, references: [] };
|
||||
const passing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 3 }) });
|
||||
await passing.run(options);
|
||||
expect(passing.prompts).toHaveLength(1);
|
||||
expect(passing.prompts).toHaveLength(SAMPLES);
|
||||
expect(passing.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
|
||||
expect(passing.records[0]).toMatchObject({ passed: true, execution: 'executed', judge_scores: { clarity: 3 } });
|
||||
const failing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 2 }) });
|
||||
@@ -254,8 +261,8 @@ test('the actual workflow callback executes once, reuses with provenance, and pr
|
||||
const f = fixture(); const first = actualCallback(f);
|
||||
const options = { ...f.opts, suite: 'Cache regression' };
|
||||
await first.run(options);
|
||||
expect(first.prompts).toEqual([f.opts.prompt]); expect(f.entries()).toHaveLength(1);
|
||||
expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 });
|
||||
expect(first.prompts).toEqual(Array(SAMPLES).fill(f.opts.prompt)); expect(f.entries()).toHaveLength(1);
|
||||
expect(first.records[0]).toMatchObject({ passed: true, execution: 'executed', cost_usd: 0.02 * SAMPLES });
|
||||
expect(first.records[0]).not.toHaveProperty('prompt');
|
||||
expect(first.records[0]).not.toHaveProperty('model');
|
||||
const reused = actualCallback(f, { judge: async () => ({ ...scores, clarity: 1 }) });
|
||||
@@ -312,13 +319,13 @@ test('a superseding attempt cancels its predecessor before either can record a s
|
||||
});
|
||||
|
||||
test('a failed input read consumes attempt one and prevents a retry from borrowing or publishing a receipt', async () => {
|
||||
const f = fixture(); f.cache().publish(scores); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8');
|
||||
const f = fixture(); f.cache().publish(panel); const receipt = fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8');
|
||||
let reads = 0;
|
||||
const h = actualCallback(f, { read: options => { if (++reads === 1) throw new Error('Missing workflow fixture'); return readWorkflowJudgeInput(options); } });
|
||||
await expect(h.run(h.options)).rejects.toThrow('Missing workflow fixture');
|
||||
expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'harness_error' });
|
||||
await h.run(h.options);
|
||||
expect(h.prompts).toHaveLength(1);
|
||||
expect(h.prompts).toHaveLength(SAMPLES);
|
||||
expect(h.records.map(record => record.execution)).toEqual(['executed', 'executed']);
|
||||
expect(h.attempts.get(f.opts.testName).attempt).toBe(2);
|
||||
expect(fs.readFileSync(path.join(f.env.EVALS_CACHE_DIR, f.entries()[0]), 'utf8')).toBe(receipt);
|
||||
@@ -333,14 +340,14 @@ test('monotonic expiry after a synchronous preparation or late model response re
|
||||
await expect(h.run(h.options)).rejects.toThrow('deadline');
|
||||
expect(h.records).toHaveLength(1);
|
||||
expect(h.records[0]).toMatchObject({ passed: false, exit_reason: 'timeout', duration_ms: 21 });
|
||||
expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : 1);
|
||||
expect(h.prompts).toHaveLength(phase === 'preparation' ? 0 : SAMPLES);
|
||||
expect(f.entries()).toHaveLength(0);
|
||||
}
|
||||
});
|
||||
|
||||
test('publication rechecks after input scanning and withdraws a receipt if recording expires', async () => {
|
||||
const f = fixture(); let checks = 0;
|
||||
f.cache().publish(scores, () => ++checks < 2);
|
||||
f.cache().publish(panel, () => ++checks < 2);
|
||||
expect(checks).toBe(2); expect(f.entries()).toHaveLength(0);
|
||||
let now = 0;
|
||||
const h = actualCallback(f, { budget: 20, allowance: 5, clock: () => now,
|
||||
@@ -373,7 +380,7 @@ test('the actual workflow callback preserves the complete public API body; cance
|
||||
try {
|
||||
const h = actualCallback(f, { judge: (prompt, model, options) => callJudge<typeof scores>(prompt, model, options) });
|
||||
await h.run(h.options);
|
||||
expect(create).toHaveBeenCalledTimes(1);
|
||||
expect(create).toHaveBeenCalledTimes(SAMPLES);
|
||||
expect(create.mock.calls[0]).toEqual([{
|
||||
model: resolveEvalModel('judge'), max_tokens: 8192,
|
||||
messages: [{ role: 'user', content: f.opts.prompt }],
|
||||
@@ -412,40 +419,40 @@ test('Ship sends its authorized 64k cap and compact response contract through th
|
||||
f.opts.structuredResponse = true;
|
||||
f.opts.maxTokens = 65_536;
|
||||
f.opts.stream = true;
|
||||
expect(f.cache().lookup()?.scores).toEqual(scores);
|
||||
expect(f.cache().lookup()?.samples).toEqual(panel);
|
||||
} finally { stream.mockRestore(); }
|
||||
});
|
||||
|
||||
test('changing response serialization misses the cache even when prompt and model match', () => {
|
||||
const f = fixture(); f.cache().publish(scores);
|
||||
const f = fixture(); f.cache().publish(panel);
|
||||
f.opts.structuredResponse = true;
|
||||
expect(f.cache().lookup()).toBeNull();
|
||||
f.cache().publish(scores);
|
||||
f.cache().publish(panel);
|
||||
expect(f.entries()).toHaveLength(2);
|
||||
expect(f.cache().lookup()?.scores).toEqual(scores);
|
||||
expect(f.cache().lookup()?.samples).toEqual(panel);
|
||||
const description = WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description;
|
||||
try {
|
||||
WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description += ' Changed response contract.';
|
||||
expect(f.cache().lookup()).toBeNull();
|
||||
} finally { WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description = description; }
|
||||
expect(f.cache().lookup()?.scores).toEqual(scores);
|
||||
expect(f.cache().lookup()?.samples).toEqual(panel);
|
||||
f.opts.structuredResponse = false;
|
||||
expect(f.cache().lookup()?.scores).toEqual(scores);
|
||||
expect(f.cache().lookup()?.samples).toEqual(panel);
|
||||
});
|
||||
|
||||
test('the actual cap and streaming transport independently affect workflow cache identity', () => {
|
||||
const f = fixture(); f.cache().publish(scores);
|
||||
const f = fixture(); f.cache().publish(panel);
|
||||
f.opts.maxTokens = 65_536;
|
||||
expect(f.cache().lookup()).toBeNull();
|
||||
f.cache().publish(scores);
|
||||
f.cache().publish(panel);
|
||||
f.opts.stream = true;
|
||||
expect(f.cache().lookup()).toBeNull();
|
||||
f.cache().publish(scores);
|
||||
f.cache().publish(panel);
|
||||
expect(f.entries()).toHaveLength(3);
|
||||
expect(f.cache().lookup()?.scores).toEqual(scores);
|
||||
expect(f.cache().lookup()?.samples).toEqual(panel);
|
||||
delete f.opts.maxTokens;
|
||||
delete f.opts.stream;
|
||||
expect(f.cache().lookup()?.scores).toEqual(scores);
|
||||
expect(f.cache().lookup()?.samples).toEqual(panel);
|
||||
});
|
||||
|
||||
test('the structured callback rejects incomplete, schema-invalid and below-threshold answers without cache credit', async () => {
|
||||
@@ -473,3 +480,96 @@ test('the structured callback rejects incomplete, schema-invalid and below-thres
|
||||
expect(validWorkflowJudgeScore({ ...scores, reasoning: Array(149).fill('word').join(' ') }, { clarity: 1, completeness: 1, actionability: 1 }, true)).toBe(true);
|
||||
} finally { stream.mockRestore(); diagnostics.mockRestore(); }
|
||||
});
|
||||
|
||||
// --- Judge panel policy (EVAL_POLICY.judge): fixed concurrent samples, per-dimension
|
||||
// mean and boolean majority against unchanged thresholds, an erroring sample fails
|
||||
// the whole panel and is never resampled. The provider is always a stub.
|
||||
const panelScore = (clarity: number, completeness = 4, actionability = 4) => ({ clarity, completeness, actionability, reasoning: `c${clarity}` });
|
||||
const panelThresholds = { clarity: 3, completeness: 3, actionability: 4 };
|
||||
const panelRefusal = () => new JudgeRefusalError({ id: 'msg_1', _request_id: 'req_1', model: 'm', usage: { input_tokens: 1, output_tokens: 0 }, content: [] });
|
||||
|
||||
describe('judge panel', () => {
|
||||
test('the pre-registered panel is three samples, and the helper restates EVAL_POLICY exactly', () => {
|
||||
expect(EVAL_POLICY.judge.samples).toBe(3);
|
||||
expect(JUDGE_PANEL_SAMPLES).toBe(EVAL_POLICY.judge.samples);
|
||||
});
|
||||
|
||||
test('draws every sample concurrently before any resolves', async () => {
|
||||
let started = 0;
|
||||
const releases: Array<() => void> = [];
|
||||
const panel = judgePanel(() => new Promise<number>(resolve => { started += 1; releases.push(() => resolve(started)); }));
|
||||
await Promise.resolve();
|
||||
expect(started).toBe(SAMPLES);
|
||||
releases.forEach(release => release());
|
||||
expect(await panel).toHaveLength(SAMPLES);
|
||||
});
|
||||
|
||||
test('an erroring sample fails the panel and is never resampled', async () => {
|
||||
let calls = 0;
|
||||
const panel = judgePanel(async () => {
|
||||
calls += 1;
|
||||
if (calls === 2) throw new Error('Judge returned non-JSON: nope');
|
||||
return panelScore(5);
|
||||
});
|
||||
await expect(panel).rejects.toThrow('non-JSON');
|
||||
expect(calls).toBe(SAMPLES);
|
||||
});
|
||||
|
||||
test('a refusal on every sample stays a provider refusal; a partial refusal is an ordinary failure', async () => {
|
||||
await expect(judgePanel(async () => { throw panelRefusal(); })).rejects.toBeInstanceOf(JudgeRefusalError);
|
||||
let calls = 0;
|
||||
const partial = judgePanel(async () => { if (++calls === 1) throw panelRefusal(); return panelScore(4); });
|
||||
const error = await partial.then(() => null, (reason: unknown) => reason);
|
||||
expect(error).toBeInstanceOf(Error);
|
||||
expect(error).not.toBeInstanceOf(JudgeRefusalError);
|
||||
expect(String(error)).toContain(`sample 1 of ${SAMPLES} failed beside scored samples`);
|
||||
});
|
||||
|
||||
test('numeric dimensions gate on the per-dimension mean; one low sample can be outvoted, a low mean cannot', () => {
|
||||
const outvoted = judgePanelMean([panelScore(2), panelScore(4), panelScore(4)], JUDGE_SCORE_DIMENSIONS);
|
||||
expect(outvoted.clarity).toBeCloseTo(10 / 3);
|
||||
expect(outvoted.clarity).toBeGreaterThanOrEqual(panelThresholds.clarity);
|
||||
const low = judgePanelMean([panelScore(2), panelScore(2), panelScore(4)], JUDGE_SCORE_DIMENSIONS);
|
||||
expect(low.clarity).toBeLessThan(panelThresholds.clarity);
|
||||
// No compensation across dimensions: each is averaged on its own.
|
||||
expect(judgePanelMean([panelScore(5, 1), panelScore(5, 1), panelScore(5, 1)], JUDGE_SCORE_DIMENSIONS).completeness).toBe(1);
|
||||
});
|
||||
|
||||
test('malformed sample fields fail the panel instead of averaging to NaN', () => {
|
||||
expect(() => judgePanelMean([panelScore(4), { ...panelScore(4), clarity: '4' as unknown as number }, panelScore(4)], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2 has non-numeric clarity');
|
||||
expect(() => judgePanelMean([panelScore(4), null as unknown as ReturnType<typeof panelScore>], JUDGE_SCORE_DIMENSIONS)).toThrow('sample 2');
|
||||
expect(() => judgePanelMean([], JUDGE_SCORE_DIMENSIONS)).toThrow('no samples');
|
||||
});
|
||||
|
||||
test('boolean fields gate on a strict majority', () => {
|
||||
const vote = (...values: boolean[]) => judgePanelMajority(values.map(value => ({ ok: value })), 'ok');
|
||||
expect(vote(true, true, false)).toBe(true);
|
||||
expect(vote(true, false, false)).toBe(false);
|
||||
expect(vote(true, false)).toBe(false);
|
||||
expect(() => judgePanelMajority([{ ok: true }, { ok: 'yes' }], 'ok')).toThrow('sample 2 has non-boolean ok');
|
||||
});
|
||||
|
||||
test('reasoning keeps every sample, numbered, even for malformed samples', () => {
|
||||
expect(judgePanelReasoning([panelScore(4), null, { reasoning: 7 }])).toBe('[sample 1] c4\n[sample 2] \n[sample 3] ');
|
||||
});
|
||||
|
||||
test('the cache stores and validates only a complete panel against the mean', () => {
|
||||
expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(true);
|
||||
expect(validWorkflowJudgePanel({ samples: [panelScore(2), panelScore(2), panelScore(4)] }, panelThresholds)).toBe(false);
|
||||
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false);
|
||||
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4), panelScore(4)] }, panelThresholds)).toBe(false);
|
||||
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), { ...panelScore(4), clarity: 6 }] }, panelThresholds)).toBe(false);
|
||||
expect(validWorkflowJudgePanel({ samples: [panelScore(4), panelScore(4), panelScore(4)], prompt: 'x' }, panelThresholds)).toBe(false);
|
||||
expect(validWorkflowJudgePanel(panelScore(4), panelThresholds)).toBe(false);
|
||||
});
|
||||
|
||||
test('every judge in the quality file samples through the panel, never a lone call', () => {
|
||||
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
|
||||
const calls = [...source.matchAll(/\b(?:callJudge<[^>(]*(?:<[^>]*>[^>(]*)*>|judge)\(/g)];
|
||||
expect(calls.length).toBeGreaterThanOrEqual(8);
|
||||
for (const call of calls) {
|
||||
expect(source.slice(Math.max(0, call.index! - 25), call.index), `unpaneled judge call at offset ${call.index}`).toMatch(/judgePanel\(\(\) => $/);
|
||||
}
|
||||
expect(source).not.toMatch(/\bscores\.reasoning\b|\bresult\.reasoning\b/);
|
||||
});
|
||||
});
|
||||
Reference in new issue
Block a user