Merge remote-tracking branch 'origin/capy/rel-a' into capy/rel-c

This commit is contained in:
garrytan committed 2026-09-29 19:47:17 +00:00
commit b2ca207cf0
46 files changed
+4936 -936

No files matched your search

+48 -100
View File
@@ -47,132 +47,80 @@ export const ALL_TIERS = {
export const SHARD_RESERVE_MS = 2 * 60_000;
/**
* Retry policy (approved 2026-09-29): a timed-out attempt is a verdict. Bun's
* --retry reruns a failed case after it may have spent its whole budget, so an
* automatic retry is kept only where one more attempt is short: every case of
* the file has a per-attempt budget of at most RETRY_MAX_CASE_MS, the CAPTURE
* tier plus its recording grace. Those failures are fast flake classes (API
* blips, tool hiccups) and a retry costs at most one more short attempt. Files
* with any longer case run once. Per-case budgets never change with this rule.
* Retry policy (approved 2026-09-29, eval reliability wave): paid evals never
* retry. Each case's kind (E2E_KINDS) fixes its trials before the run: `rule`
* one trial, `behavior` a panel of EVAL_POLICY.panel independent trials, and
* `judge` one case that samples its judge panel internally. A failed verdict
* is final for that run; a manual re-run adds trials under a new run attempt
* and never replaces the original verdict. Rows below keep only wall
* supervision; per-case budgets never change with this rule.
*/
export const RETRY_MAX_CASE_MS = CAPTURE_MS + 15_000;
export function retriesWithinCaseCap(caseMs: number, configuredRetries: number): number {
return caseMs <= RETRY_MAX_CASE_MS ? configuredRetries : 0;
}
/**
* Unregistered paid files that keep one automatic retry: every case budget is
* JUDGE or CAPTURE tier (test/paid-retry-supervision.test.ts scans each source).
* Registered rows below derive retries from their declared caseMs; every other
* paid file runs once.
*/
export const SHORT_CASE_RETRY_FILES: readonly string[] = [
'test/codex-e2e-sol-scope.test.ts',
'test/llm-judge-recommendation.test.ts',
'test/skill-e2e-ask-user-question-format-compliance.test.ts',
'test/skill-e2e-benchmark-providers.test.ts',
'test/skill-e2e-bws.test.ts',
'test/skill-e2e-context-skills.test.ts',
'test/skill-e2e-coverage-audit.test.ts',
'test/skill-e2e-diagram.test.ts',
'test/skill-e2e-first-task-scaffold.test.ts',
'test/skill-e2e-gbrain-roundtrip-local.test.ts',
'test/skill-e2e-hermetic-canary.test.ts',
'test/skill-e2e-investigate-owned-completion.test.ts',
'test/skill-e2e-investigate-owned-termination.test.ts',
'test/skill-e2e-learnings.test.ts',
'test/skill-e2e-plan-tune.test.ts',
'test/skill-e2e-qa-functional-fix.test.ts',
'test/skill-e2e-qa-functional.test.ts',
'test/skill-e2e-review-army.test.ts',
'test/skill-e2e-review.test.ts',
'test/skill-e2e-session-intelligence.test.ts',
'test/skill-e2e-setup-gbrain-bad-token.test.ts',
'test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts',
'test/skill-e2e-setup-gbrain-remote.test.ts',
'test/skill-e2e-ship-hook-consent.test.ts',
'test/skill-e2e-ship-hook-refresh.test.ts',
'test/skill-e2e-ship-skip.test.ts',
'test/skill-e2e-sync-gbrain-readiness.test.ts',
'test/skill-e2e-third-party-actions.test.ts',
'test/skill-e2e-triage.test.ts',
'test/skill-routing-e2e.test.ts',
];
/** Whole-file supervision covers every attempt the retry policy allows.
* These fixtures allow 25 minutes per case, so they run once.
/** Whole-file supervision for one run of every case.
* These fixtures allow 25 minutes per case.
* Reserve the sequential upper bound even when Bun runs sibling cases together.
*/
export const FINDING_RETRY_BUDGETS = [
{ file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 },
].map(({ file, cases }) => {
const retries = retriesWithinCaseCap(1_500_000, 1);
return {
file, cases,
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
testMs: 1_500_000,
caseMs: 1_500_000,
retries,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: cases * 1_500_000 * (retries + 1) + SHARD_RESERVE_MS,
};
});
].map(({ file, cases }) => ({
file, cases,
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
testMs: 1_500_000,
caseMs: 1_500_000,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: cases * 1_500_000 + SHARD_RESERVE_MS,
}));
/** Three existing captures in one 16-minute case, so the file runs once. */
/** Three existing captures in one 16-minute case. */
export const AUQ_CONSISTENCY_RETRY_BUDGET = {
file: 'test/skill-e2e-auq-consistency.test.ts',
id: 'auq-consistency-existing-retry-v1',
cases: 1,
testMs: 3 * CAPTURE_MS + 60_000,
caseMs: 3 * CAPTURE_MS + 60_000,
retries: retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1),
shardReserveMs: SHARD_RESERVE_MS,
shardMs: (3 * CAPTURE_MS + 60_000) * (retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1) + 1) + SHARD_RESERVE_MS,
shardMs: 3 * CAPTURE_MS + 60_000 + SHARD_RESERVE_MS,
} as const;
/** These fixtures have a fixed case count in every supported tier. */
export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET];
/** Whole-file walls cover all existing cases and every allowed attempt, even if
* Bun runs them sequentially. Mixed-tier files reserve their larger complete
* tier, never a currently selected subset. caseMs is the longest single case
* budget, which decides the retry (RETRY_MAX_CASE_MS). These rows add no
* case-count or model-work policy. The 10-second terms preserve the existing
* Codex/recording finalization grace.
/** Whole-file walls cover all existing cases, even if Bun runs them
* sequentially. Mixed-tier files reserve their larger complete tier, never a
* currently selected subset. caseMs is the longest single case budget, the
* wall of one isolated case shard. These rows add no case-count or model-work
* policy. The 10-second terms preserve the existing Codex/recording
* finalization grace.
*/
export const FILE_RETRY_BUDGETS = [
...STRICT_RETRY_CASE_BUDGETS,
...[
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000 },
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS },
// Seventeen workflow judges include their 10s recording grace; the other
// seven judges retain 120s. Supervise all 24 and the existing one retry.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 2 },
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
// seven judges retain 120s. Supervise all 24.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 },
// Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s.
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS, configuredRetries: 1 },
].map(({ file, attemptMs, caseMs, configuredRetries }) => {
const retries = retriesWithinCaseCap(caseMs, configuredRetries);
return {
file, attemptMs, caseMs, retries,
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
};
}),
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS },
].map(({ file, attemptMs, caseMs }) => ({
file, attemptMs, caseMs,
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: attemptMs + SHARD_RESERVE_MS,
})),
];
/** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */
+201 -1
View File
@@ -14,8 +14,20 @@ import {
formatComparison,
generateCommentary,
judgePassed,
CONTRACT_VIOLATIONS_FILE,
ContractViolation,
TRIAL_ENV,
TRIAL_OUTCOME_SCHEMA,
expectContract,
failureClassOf,
formatTrialOutcomes,
panelVerdict,
parseTrialOutcomes,
sanitizeTrialError,
trialContextFromEnv,
} from './eval-store';
import type { EvalResult, EvalTestEntry, ComparisonResult } from './eval-store';
import type { EvalResult, EvalTestEntry, ComparisonResult, PanelTrial, TrialOutcomeRecord } from './eval-store';
import { EVAL_POLICY } from './periodic-exclude-data';
import { manualReviewFixture } from './manual-judge-review-fixture';
let tmpDir: string;
@@ -957,3 +969,191 @@ describe('generateCommentary', () => {
expect(notes.some(n => n.includes('Stable run'))).toBe(true);
});
});
// --- Trials, panel verdicts and contract vetoes (eval reliability policy) ---
const PANEL = EVAL_POLICY.panel;
const pass = (trial: number, extra: Partial<PanelTrial> = {}): PanelTrial => ({ trial, outcome: 'passed', ...extra });
const fail = (trial: number, extra: Partial<PanelTrial> = {}): PanelTrial => ({ trial, outcome: 'failed', ...extra });
const behavior = (trials: PanelTrial[], quarantined = false) =>
panelVerdict({ case: 'case-x', kind: 'behavior', panel: PANEL, trials, quarantined });
describe('panelVerdict', () => {
test('policy constants are the approved pre-registration', () => {
expect(EVAL_POLICY.panel).toEqual({ n: 3, k: 2 });
expect(EVAL_POLICY.quarantine).toEqual({
entry: { rate: 0.95, minTrials: 10 },
exit: { rate: 0.97, minTrials: 10 },
capFraction: 0.1,
expiryWeeklyRuns: 8,
});
expect(EVAL_POLICY.infraRedispatch).toBe(1);
});
test('rule: one trial, any failure fails the lane', () => {
const ok = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 1, k: 1 }, trials: [pass(1)] });
expect(ok).toMatchObject({ status: 'PASS', split: false, failsLane: false, coverage: true, marks: '✓' });
const bad = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 1, k: 1 }, trials: [fail(1)] });
expect(bad).toMatchObject({ status: 'FAIL', failsLane: true, coverage: false, redClass: 'VERDICT', marks: '✗' });
});
test('behavior 3/3 is a clean PASS', () => {
expect(behavior([pass(1), pass(2), pass(3)])).toMatchObject({ status: 'PASS', split: false, passed: 3, failsLane: false });
});
test('behavior 2/3 is a split PASS that shows its failed trial', () => {
const v = behavior([pass(1), fail(2, { exit_reason: 'timeout' }), pass(3)]);
expect(v).toMatchObject({ status: 'PASS', split: true, passed: 2, failed: 1, failsLane: false, coverage: true, marks: '✓✗✓', reason: 'PASS 2/3' });
expect(v.trials[1].exit_reason).toBe('timeout');
});
test('behavior 1/3 and 0/3 fail the lane', () => {
expect(behavior([pass(1), fail(2), fail(3)])).toMatchObject({ status: 'FAIL', failsLane: true, redClass: 'VERDICT' });
expect(behavior([fail(1), fail(2), fail(3)])).toMatchObject({ status: 'FAIL', failsLane: true });
});
test('a contract trial fails the panel even at 2/3', () => {
const v = behavior([pass(1), pass(2), fail(3, { failure_class: 'contract' })]);
expect(v).toMatchObject({ status: 'FAIL', contract: true, failsLane: true, redClass: 'VERDICT', reason: 'contract violation' });
});
test('a missing trial is INCOMPLETE and fails the lane', () => {
const v = behavior([pass(1), pass(3)]);
expect(v).toMatchObject({ status: 'INCOMPLETE', failsLane: true, coverage: false, redClass: 'INCOMPLETE', marks: '✓·✓' });
expect(v.reason).toContain('missing trial t2');
});
test('duplicate or out-of-range trial records are INCOMPLETE, never deduplicated', () => {
expect(behavior([pass(1), pass(2), pass(2), fail(3)]).status).toBe('INCOMPLETE');
expect(behavior([pass(1), pass(2), pass(3), pass(4)]).reason).toContain('unexpected trial t4');
const v = panelVerdict({ case: 'c', kind: 'behavior', panel: { n: 11, k: 6 }, trials: Array.from({ length: 10 }, (_, i) => pass(i + 2)) });
expect(v.reason).toContain('missing trial t1');
});
test('timeout and infra trials count as failed, never passing', () => {
const v = behavior([pass(1), fail(2, { exit_reason: 'timeout' }), fail(3, { failure_class: 'infra' })]);
expect(v).toMatchObject({ status: 'FAIL', passed: 1, failed: 2, failsLane: true, redClass: 'VERDICT' });
const infra = behavior([pass(1), fail(2, { failure_class: 'infra' }), fail(3, { failure_class: 'infra' })]);
expect(infra).toMatchObject({ status: 'FAIL', redClass: 'INFRA' });
expect(failureClassOf({ exit_reason: 'timeout' })).toBe('timeout');
expect(failureClassOf({})).toBe('assertion');
});
test('quarantined: 1/3 does not fail the lane, 0/3 and contract do, no coverage credit', () => {
expect(behavior([pass(1), fail(2), fail(3)], true)).toMatchObject({ status: 'FAIL', failsLane: false, coverage: false, redClass: null });
expect(behavior([fail(1), fail(2), fail(3)], true)).toMatchObject({ status: 'FAIL', failsLane: true });
expect(behavior([pass(1), pass(2), fail(3, { failure_class: 'contract' })], true)).toMatchObject({ status: 'FAIL', failsLane: true });
expect(behavior([pass(1), pass(2), pass(3)], true)).toMatchObject({ status: 'PASS', coverage: false, failsLane: false });
expect(behavior([pass(1), pass(2)], true)).toMatchObject({ status: 'INCOMPLETE', failsLane: true });
});
test('quarantined rule keeps rule meaning (k = n)', () => {
const v = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 3, k: 3 }, trials: [pass(1), pass(2), fail(3)], quarantined: true });
expect(v).toMatchObject({ status: 'FAIL', failsLane: false });
});
test('all-skipped panel is SKIPPED with no credit; partly skipped is INCOMPLETE', () => {
const skip = (trial: number): PanelTrial => ({ trial, outcome: 'skipped' });
expect(behavior([skip(1), skip(2), skip(3)])).toMatchObject({ status: 'SKIPPED', coverage: false, failsLane: false });
expect(behavior([pass(1), pass(2), skip(3)])).toMatchObject({ status: 'INCOMPLETE', failsLane: true });
});
test('trials of different run attempts are never merged into one verdict', () => {
expect(() => behavior([pass(1), pass(2), fail(3, { attempt: 2 })])).toThrow(/run attempts/);
expect(behavior([pass(1, { attempt: 2 }), pass(2, { attempt: 2 }), pass(3, { attempt: 2 })]).attempt).toBe(2);
});
test('invalid panels throw', () => {
expect(() => panelVerdict({ case: 'c', kind: 'behavior', panel: { n: 3, k: 4 }, trials: [] })).toThrow(/invalid panel/);
expect(() => panelVerdict({ case: 'c', kind: 'nope' as any, panel: { n: 1, k: 1 }, trials: [] })).toThrow(/unknown kind/);
});
});
describe('trial context and expectContract', () => {
let dir: string;
const saved: Record<string, string | undefined> = {};
const keys = [...Object.values(TRIAL_ENV), 'GSTACK_EVAL_DIR'];
beforeEach(() => {
dir = fs.mkdtempSync(path.join(os.tmpdir(), 'panel-verdict-'));
for (const key of keys) saved[key] = process.env[key];
});
afterEach(() => {
for (const key of keys) {
if (saved[key] === undefined) delete process.env[key];
else process.env[key] = saved[key];
}
fs.rmSync(dir, { recursive: true, force: true });
});
const setTrial = () => Object.assign(process.env, {
[TRIAL_ENV.caseId]: 'case-x', [TRIAL_ENV.kind]: 'behavior', [TRIAL_ENV.trial]: '2',
[TRIAL_ENV.panelN]: '3', [TRIAL_ENV.panelK]: '2', [TRIAL_ENV.policyVersion]: String(EVAL_POLICY.version),
GSTACK_EVAL_DIR: dir,
});
test('trialContextFromEnv: absent, complete, and malformed', () => {
for (const key of Object.values(TRIAL_ENV)) delete process.env[key];
expect(trialContextFromEnv()).toBeNull();
setTrial();
expect(trialContextFromEnv()).toEqual({ case_id: 'case-x', kind: 'behavior', trial: 2, panel: { n: 3, k: 2 }, policy_version: EVAL_POLICY.version });
process.env[TRIAL_ENV.trial] = '4';
expect(() => trialContextFromEnv()).toThrow(/Malformed trial context/);
});
test('passing contract is a no-op', () => {
setTrial();
expect(() => expectContract(true, 'fine')).not.toThrow();
expect(fs.existsSync(path.join(dir, CONTRACT_VIOLATIONS_FILE))).toBe(false);
});
test('failed contract stamps the recorded entry and the sidecar before throwing', () => {
setTrial();
const collector = new EvalCollector('e2e', dir);
collector.addTest({ name: 'case-x', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 });
expect(() => expectContract(false, 'handoff missing', { collector, name: 'case-x' })).toThrow(ContractViolation);
const partial = JSON.parse(fs.readFileSync(path.join(dir, '_partial-e2e.json'), 'utf-8'));
expect(partial.tests[0]).toMatchObject({ passed: false, failure_class: 'contract', case_id: 'case-x', trial: 2, kind: 'behavior', panel: { n: 3, k: 2 } });
const sidecar = fs.readFileSync(path.join(dir, CONTRACT_VIOLATIONS_FILE), 'utf-8').trim().split('\n').map((l) => JSON.parse(l));
expect(sidecar).toEqual([expect.objectContaining({ case_id: 'case-x', trial: 2, message: 'handoff missing' })]);
});
test('a contract marked before recording stamps the later record, or becomes its own at finalize', async () => {
setTrial();
const collector = new EvalCollector('e2e', dir);
expect(() => expectContract(0, 'no question asked', { collector, name: 'later' })).toThrow('CONTRACT: no question asked');
collector.addTest({ name: 'later', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 });
expect(() => expectContract(null, 'never recorded', { collector, name: 'orphan' })).toThrow();
const file = await collector.finalize();
const tests = JSON.parse(fs.readFileSync(file, 'utf-8')).tests;
expect(tests.find((t: any) => t.name === 'later')).toMatchObject({ passed: false, failure_class: 'contract' });
expect(tests.find((t: any) => t.name === 'orphan')).toMatchObject({ passed: false, failure_class: 'contract', error: 'never recorded' });
});
});
describe('trial-outcomes JSONL', () => {
const record = (extra: Partial<TrialOutcomeRecord> = {}): TrialOutcomeRecord => ({
schema: TRIAL_OUTCOME_SCHEMA, case: 'case-x', file: 'test/x.test.ts', tier: 'gate', kind: 'behavior',
trial: 1, panel: { n: 3, k: 2 }, attempt: 1, outcome: 'passed', duration_ms: 10, cost_usd: 0.1,
policy_version: EVAL_POLICY.version, quarantined: false, execution: 'executed', source: 'shard', ...extra,
});
test('round-trips valid records', () => {
const records = [record(), record({ trial: 2, outcome: 'failed', failure_class: 'timeout', exit_reason: 'timeout' })];
expect(parseTrialOutcomes(formatTrialOutcomes(records))).toEqual({ records, errors: [] });
});
test('writer fails closed; reader reports bad lines as data errors', () => {
expect(() => formatTrialOutcomes([record({ outcome: 'failed' })])).toThrow(/failed without failure_class/);
expect(() => formatTrialOutcomes([record({ trial: 4 })])).toThrow(/trial invalid/);
const text = `${JSON.stringify(record())}\nnot json\n${JSON.stringify({ ...record(), schema: 'other' })}\n`;
const parsed = parseTrialOutcomes(text);
expect(parsed.records).toHaveLength(1);
expect(parsed.errors).toEqual(['line 2: not JSON', 'line 3: schema other']);
expect(parseTrialOutcomes(text, { maxBytes: 10 }).errors[0]).toContain('exceed');
});
test('sanitizeTrialError keeps one capped line without mentions', () => {
expect(sanitizeTrialError('\n expected @garrytan to `see`\nsecond')).toBe("expected @\u200bgarrytan to 'see'");
expect(sanitizeTrialError('x'.repeat(1000))!.length).toBe(300);
expect(sanitizeTrialError('')).toBeUndefined();
});
});
+367 -1
View File
@@ -76,6 +76,18 @@ export interface EvalTestEntry {
* its body again and re-records under the same name. Set by addTest. */
attempt?: number;
// Trial identity (eval reliability policy). Stamped by addTest from the
// TRIAL_ENV variables the paid runner sets on an isolated trial shard.
/** Registry id (E2E_TIERS / LLM_JUDGE_TOUCHFILES key) this record belongs to. */
case_id?: string;
kind?: EvalCaseKind;
/** 1-based trial index within the case's panel. */
trial?: number;
panel?: PanelShape;
/** Why a failed record failed; 'contract' comes only from expectContract. */
failure_class?: TrialFailureClass;
policy_version?: number;
// E2E
transcript?: any[];
prompt?: string;
@@ -131,6 +143,329 @@ export function evalEntryOutcome(entry: unknown): 'passed' | 'failed' | 'manual-
return result.passed === true ? 'passed' : 'failed';
}
// --- Trials and panel verdicts ---
//
// Paid evals never retry. Each case's kind (E2E_KINDS) fixes its trials before
// the run; a panel verdict is computed once, by panelVerdict(), from exactly
// panel.n trial records of one run attempt. The report, collector-outcomes,
// the PR comment and pass-rates all read that one function.
export type EvalCaseKind = 'rule' | 'behavior' | 'judge';
/** assertion: an ordinary failed expectation. contract: expectContract() fired
* (fails the panel at any count). timeout: the case budget ran out.
* infra: API/CLI/runner failure before the model could be graded. */
export type TrialFailureClass = 'assertion' | 'contract' | 'timeout' | 'infra';
export type TrialOutcome = 'passed' | 'failed' | 'skipped';
export interface PanelShape { n: number; k: number }
/** Environment the paid runner sets on an isolated trial shard. */
export const TRIAL_ENV = {
caseId: 'GSTACK_EVAL_CASE_ID',
kind: 'GSTACK_EVAL_KIND',
trial: 'GSTACK_EVAL_TRIAL',
panelN: 'GSTACK_EVAL_PANEL_N',
panelK: 'GSTACK_EVAL_PANEL_K',
policyVersion: 'GSTACK_EVAL_POLICY_VERSION',
} as const;
/** Sidecar every expectContract() failure appends to (in GSTACK_EVAL_DIR), so a
* contract veto survives a test that throws before recording its entry. */
export const CONTRACT_VIOLATIONS_FILE = 'contract-violations.jsonl';
export interface TrialContext {
case_id: string;
kind: EvalCaseKind;
trial: number;
panel: PanelShape;
policy_version: number;
}
const EVAL_KINDS: readonly EvalCaseKind[] = ['rule', 'behavior', 'judge'];
const FAILURE_CLASSES: readonly TrialFailureClass[] = ['assertion', 'contract', 'timeout', 'infra'];
function positiveInt(raw: string | undefined): number | null {
if (raw === undefined || !/^[1-9][0-9]*$/.test(raw)) return null;
return Number(raw);
}
/** Trial context of this process, or null outside an isolated trial shard.
* A partial or malformed context throws: a mislabeled trial is fail-open. */
export function trialContextFromEnv(env: NodeJS.ProcessEnv = process.env): TrialContext | null {
const caseId = env[TRIAL_ENV.caseId];
if (!caseId) return null;
const kind = env[TRIAL_ENV.kind] as EvalCaseKind | undefined;
const trial = positiveInt(env[TRIAL_ENV.trial]);
const n = positiveInt(env[TRIAL_ENV.panelN]);
const k = positiveInt(env[TRIAL_ENV.panelK]);
const policy = positiveInt(env[TRIAL_ENV.policyVersion]);
if (!kind || !EVAL_KINDS.includes(kind) || trial === null || n === null || k === null || policy === null || k > n || trial > n) {
throw new Error(`Malformed trial context for ${caseId}: ${Object.values(TRIAL_ENV).map((name) => `${name}=${env[name] ?? ''}`).join(' ')}`);
}
return { case_id: caseId, kind, trial, panel: { n, k }, policy_version: policy };
}
/** Failure class of a failed record: an explicit class wins, then the exit reason. */
export function failureClassOf(entry: Pick<EvalTestEntry, 'failure_class' | 'exit_reason'>): TrialFailureClass {
if (entry.failure_class && FAILURE_CLASSES.includes(entry.failure_class)) return entry.failure_class;
return entry.exit_reason === 'timeout' ? 'timeout' : 'assertion';
}
export class ContractViolation extends Error {
constructor(message: string) {
super(`CONTRACT: ${message}`);
this.name = 'ContractViolation';
}
}
/**
* Assert a contract: an outcome the product must meet on every run. On failure
* it records failure_class 'contract' before throwing, both on the collector
* entry named `record.name` (now or when the test records it) and in the
* GSTACK_EVAL_DIR sidecar, so panelVerdict() fails the panel even at 2 of 3.
*/
export function expectContract(
condition: unknown,
message: string,
record?: { collector: EvalCollector | null; name: string },
): asserts condition {
if (condition) return;
record?.collector?.markContractViolation(record.name, message);
const evalDir = process.env.GSTACK_EVAL_DIR;
if (evalDir) {
const context = trialContextFromEnv();
fs.mkdirSync(evalDir, { recursive: true });
fs.appendFileSync(path.join(evalDir, CONTRACT_VIOLATIONS_FILE), JSON.stringify({
case_id: context?.case_id ?? record?.name ?? null,
name: record?.name ?? null,
trial: context?.trial ?? null,
message,
at: new Date().toISOString(),
}) + '\n');
}
throw new ContractViolation(message);
}
export interface PanelTrial {
trial: number;
outcome: TrialOutcome;
/** Required meaning for a failed trial; absent reads as 'assertion'. */
failure_class?: TrialFailureClass;
/** CI run attempt (github.run_attempt); absent means 1. */
attempt?: number;
exit_reason?: string;
error?: string;
execution?: 'executed' | 'reused';
}
export interface PanelVerdictInput {
case: string;
kind: EvalCaseKind;
panel: PanelShape;
trials: readonly PanelTrial[];
quarantined?: boolean;
}
export type PanelStatus = 'PASS' | 'FAIL' | 'INCOMPLETE' | 'SKIPPED';
export interface PanelVerdict {
case: string;
kind: EvalCaseKind;
panel: PanelShape;
attempt: number;
quarantined: boolean;
status: PanelStatus;
passed: number;
failed: number;
/** A failed trial carried failure_class 'contract'. */
contract: boolean;
/** PASS with at least one failed trial: shown as `PASS k/n`, never clean. */
split: boolean;
/** Whether this verdict makes the lane red. */
failsLane: boolean;
/** Whether it counts as passing coverage (never for quarantined or skipped). */
coverage: boolean;
/** Machine classification of a lane-failing verdict: INCOMPLETE (missing or
* malformed trial records), INFRA (every failed trial is infra-class), or
* VERDICT (a real red). Null when the verdict does not fail the lane. */
redClass: 'INCOMPLETE' | 'INFRA' | 'VERDICT' | null;
/** One glyph per trial index: ✓ pass, ✗ fail, – skipped, · missing. */
marks: string;
reason: string;
trials: PanelTrial[];
}
/**
* The single verdict function. `rule`/`judge` cases run panel {1,1}; `behavior`
* cases run EVAL_POLICY.panel; a quarantined case runs a full panel whose k
* keeps its kind's meaning (k = n for rule). Verdict: INCOMPLETE unless
* exactly one record per trial index 1..n; SKIPPED when every trial skipped;
* FAIL on any contract trial; otherwise PASS iff passed >= k. A quarantined
* FAIL fails the lane only on a hard break (0 of n) or a contract violation.
*/
export function panelVerdict(input: PanelVerdictInput): PanelVerdict {
const { n, k } = input.panel;
if (!Number.isInteger(n) || !Number.isInteger(k) || n < 1 || k < 1 || k > n) {
throw new Error(`${input.case}: invalid panel {n:${n}, k:${k}}`);
}
if (!EVAL_KINDS.includes(input.kind)) throw new Error(`${input.case}: unknown kind ${String(input.kind)}`);
const attempts = new Set(input.trials.map((t) => t.attempt ?? 1));
if (attempts.size > 1) {
throw new Error(`${input.case}: trials from run attempts ${[...attempts].join(', ')}; compute one verdict per attempt`);
}
const attempt = [...attempts][0] ?? 1;
const quarantined = input.quarantined === true;
const trials = [...input.trials].sort((a, b) => a.trial - b.trial);
const byIndex = new Map<number, PanelTrial>();
const problems: string[] = [];
for (const t of trials) {
if (!Number.isInteger(t.trial) || t.trial < 1 || t.trial > n) problems.push(`unexpected trial t${t.trial}`);
else if (byIndex.has(t.trial)) problems.push(`duplicate trial t${t.trial}`);
else if (t.outcome !== 'passed' && t.outcome !== 'failed' && t.outcome !== 'skipped') problems.push(`t${t.trial} has outcome ${String(t.outcome)}`);
else byIndex.set(t.trial, t);
}
for (let i = 1; i <= n; i++) if (!trials.some((t) => t.trial === i)) problems.push(`missing trial t${i}`);
const marks = Array.from({ length: n }, (_, i) => {
const t = byIndex.get(i + 1);
return !t ? '·' : t.outcome === 'passed' ? '✓' : t.outcome === 'failed' ? '✗' : '–';
}).join('');
const passed = [...byIndex.values()].filter((t) => t.outcome === 'passed').length;
const failedTrials = [...byIndex.values()].filter((t) => t.outcome === 'failed');
const skipped = [...byIndex.values()].filter((t) => t.outcome === 'skipped').length;
const contract = failedTrials.some((t) => failureClassOf(t) === 'contract');
const base = { case: input.case, kind: input.kind, panel: { n, k }, attempt, quarantined, passed, failed: failedTrials.length, contract, marks, trials };
if (problems.length === 0 && skipped === n) {
return { ...base, status: 'SKIPPED', split: false, failsLane: false, coverage: false, redClass: null, reason: 'every trial skipped (no verdict credit)' };
}
if (problems.length === 0 && skipped > 0) problems.push(`${skipped} of ${n} trials skipped`);
if (problems.length > 0) {
return { ...base, status: 'INCOMPLETE', split: false, failsLane: true, coverage: false, redClass: 'INCOMPLETE', reason: problems.join('; ') };
}
if (!contract && passed >= k) {
const split = failedTrials.length > 0;
return {
...base, status: 'PASS', split, failsLane: false, coverage: !quarantined, redClass: null,
reason: split ? `PASS ${passed}/${n}` : `${passed}/${n} passed`,
};
}
const hardBreak = passed === 0;
const failsLane = !quarantined || contract || hardBreak;
const allInfra = !contract && failedTrials.length > 0 && failedTrials.every((t) => failureClassOf(t) === 'infra');
const why = contract ? 'contract violation' : `${passed}/${n} passed, needs ${k}`;
return {
...base, status: 'FAIL', split: false, failsLane, coverage: false,
redClass: failsLane ? (allInfra ? 'INFRA' : 'VERDICT') : null,
reason: !quarantined ? why
: contract ? `${why}; quarantine never excuses a contract`
: hardBreak ? `${why}; quarantined hard break`
: `${why}; quarantined, does not fail the lane`,
};
}
// --- trial-outcomes JSONL (one line per trial; pass-rate history input) ---
export const TRIAL_OUTCOME_SCHEMA = 'gstack-trial-outcome/v1';
export const TRIAL_OUTCOMES_FILE = 'trial-outcomes.jsonl';
/** Cap on a stored `error` line (sanitized first line of the failure). */
export const TRIAL_ERROR_MAX = 300;
export interface TrialOutcomeRecord {
schema: typeof TRIAL_OUTCOME_SCHEMA;
/** Registry id. */
case: string;
file: string;
tier: string;
kind: EvalCaseKind;
trial: number;
panel: PanelShape;
/** CI run attempt (github.run_attempt); 1 locally and for pre-policy backfill. */
attempt: number;
outcome: TrialOutcome;
/** Present exactly when outcome is 'failed'. */
failure_class?: TrialFailureClass;
exit_reason?: string;
error?: string;
duration_ms: number;
cost_usd: number;
model?: string;
cli_version?: string;
/** Reuse input key of the trial's shard, when known. */
input_identity?: string;
/** EVAL_POLICY.version; 0 marks pre-policy backfill. */
policy_version: number;
quarantined: boolean;
execution: 'executed' | 'reused';
/** shard: isolated trial shard status. junit: a rule file shard's per-test
* JUnit outcome. backfill: imported pre-policy artifact record. */
source: 'shard' | 'junit' | 'backfill';
run_id?: string;
sha?: string;
lane?: string;
recorded_at?: string;
/** History series key: a hash of the case's own touchfiles (GLOBAL_TOUCHFILES excluded), stamped by the report job. */
series_identity?: string;
}
/** First line of free text, stripped of @-mentions and control characters, capped. */
export function sanitizeTrialError(text: string | undefined): string | undefined {
if (!text) return undefined;
const first = text.split('\n').map((l) => l.trim()).find((l) => l.length > 0);
if (!first) return undefined;
// eslint-disable-next-line no-control-regex
const clean = first.replace(/[\u0000-\u001f\u007f]/g, ' ').replace(/`/g, "'").replace(/@(?=[A-Za-z0-9_-])/g, '@\u200b');
return clean.length > TRIAL_ERROR_MAX ? `${clean.slice(0, TRIAL_ERROR_MAX - 1)}…` : clean;
}
function trialRecordProblems(r: any): string[] {
const problems: string[] = [];
if (!r || typeof r !== 'object' || Array.isArray(r)) return ['not an object'];
if (r.schema !== TRIAL_OUTCOME_SCHEMA) problems.push(`schema ${String(r.schema)}`);
for (const key of ['case', 'file', 'tier'] as const) if (typeof r[key] !== 'string' || r[key].length === 0) problems.push(`${key} missing`);
if (!EVAL_KINDS.includes(r.kind)) problems.push(`kind ${String(r.kind)}`);
const n = r.panel?.n, k = r.panel?.k;
if (!Number.isInteger(n) || !Number.isInteger(k) || n < 1 || k < 1 || k > n) problems.push('panel invalid');
if (!Number.isInteger(r.trial) || r.trial < 1 || (Number.isInteger(n) && r.trial > n)) problems.push('trial invalid');
if (!Number.isInteger(r.attempt) || r.attempt < 1) problems.push('attempt invalid');
if (!['passed', 'failed', 'skipped'].includes(r.outcome)) problems.push(`outcome ${String(r.outcome)}`);
if (r.outcome === 'failed' && !FAILURE_CLASSES.includes(r.failure_class)) problems.push('failed without failure_class');
if (r.outcome !== 'failed' && r.failure_class !== undefined) problems.push('failure_class on a non-failed trial');
if (typeof r.duration_ms !== 'number' || !Number.isFinite(r.duration_ms) || r.duration_ms < 0) problems.push('duration_ms invalid');
if (typeof r.cost_usd !== 'number' || !Number.isFinite(r.cost_usd) || r.cost_usd < 0) problems.push('cost_usd invalid');
if (!Number.isInteger(r.policy_version) || r.policy_version < 0) problems.push('policy_version invalid');
if (typeof r.quarantined !== 'boolean') problems.push('quarantined invalid');
if (r.execution !== 'executed' && r.execution !== 'reused') problems.push('execution invalid');
if (!['shard', 'junit', 'backfill'].includes(r.source)) problems.push('source invalid');
if (r.error !== undefined && (typeof r.error !== 'string' || r.error.length > TRIAL_ERROR_MAX)) problems.push('error invalid');
if (r.series_identity !== undefined && (typeof r.series_identity !== 'string' || !/^[\w.-]{1,64}$/.test(r.series_identity))) problems.push('series_identity invalid');
return problems;
}
/** Serialize records as JSONL; throws on any invalid record (writers fail closed). */
export function formatTrialOutcomes(records: readonly TrialOutcomeRecord[]): string {
return records.map((r) => {
const problems = trialRecordProblems(r);
if (problems.length > 0) throw new Error(`invalid trial record ${r?.case}~t${r?.trial}: ${problems.join(', ')}`);
return JSON.stringify(r);
}).join('\n') + (records.length > 0 ? '\n' : '');
}
/** Parse downloaded JSONL as data only: invalid lines are reported, never guessed. */
export function parseTrialOutcomes(text: string, opts: { maxBytes?: number } = {}): { records: TrialOutcomeRecord[]; errors: string[] } {
const maxBytes = opts.maxBytes ?? 16 * 1024 * 1024;
if (Buffer.byteLength(text) > maxBytes) return { records: [], errors: [`trial outcomes exceed ${maxBytes} bytes`] };
const records: TrialOutcomeRecord[] = [];
const errors: string[] = [];
text.split('\n').forEach((line, i) => {
if (line.trim() === '') return;
let parsed: unknown;
try { parsed = JSON.parse(line); } catch { errors.push(`line ${i + 1}: not JSON`); return; }
const problems = trialRecordProblems(parsed);
if (problems.length > 0) errors.push(`line ${i + 1}: ${problems.join(', ')}`);
else records.push(parsed as TrialOutcomeRecord);
});
return { records, errors };
}
export interface EvalResult {
schema_version: number;
version: string;
@@ -887,6 +1222,7 @@ export class EvalCollector {
private shard: string | null;
private fileNamespace?: string;
private createdAt = Date.now();
private pendingContract = new Map<string, string>();
constructor(tier: 'e2e' | 'llm-judge', evalDir?: string, fileNamespace?: string) {
if (fileNamespace !== undefined && !/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(fileNamespace)) {
@@ -903,7 +1239,29 @@ export class EvalCollector {
// names are unique by convention). Stamp the 1-based attempt so a
// pass-on-attempt-2 stays visible forever — the stream hides it.
const prior = this.tests.filter((t) => t.name === entry.name).length;
this.tests.push({ ...entry, attempt: prior + 1 });
const context = trialContextFromEnv();
const record: EvalTestEntry = { ...(context ?? {}), ...entry, attempt: prior + 1 };
const contract = this.pendingContract.get(entry.name);
if (contract !== undefined) {
this.pendingContract.delete(entry.name);
Object.assign(record, { passed: false, failure_class: 'contract', error: record.error ?? contract });
}
this.tests.push(record);
this.savePartial();
}
/** expectContract() hook: mark `name`'s latest record (or its next one) as a
* contract failure. An unmatched mark becomes its own failed record at
* finalize, so the veto is never lost. */
markContractViolation(name: string, message: string): void {
const existing = this.tests.filter((t) => t.name === name).at(-1);
if (!existing) {
this.pendingContract.set(name, message);
return;
}
existing.passed = false;
existing.failure_class = 'contract';
existing.error = existing.error ?? message;
this.savePartial();
}
@@ -959,6 +1317,14 @@ export class EvalCollector {
async finalize(): Promise<string> {
if (this.finalized) return '';
this.finalized = true;
for (const [name, message] of this.pendingContract) {
this.tests.push({
...(trialContextFromEnv() ?? {}),
name, suite: 'contract', tier: this.tier, passed: false, duration_ms: 0, cost_usd: 0,
failure_class: 'contract', error: message, attempt: 1,
});
}
this.pendingContract.clear();
const git = getGitInfo();
const version = getVersion();
+59
View File
@@ -23,6 +23,8 @@ export interface JudgeScore {
reasoning: string;
}
export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const;
export interface JudgeRefusalEvidence {
stop_reason: 'refusal';
response_id: string | null;
@@ -196,6 +198,63 @@ export async function callJudge<T>(
}
}
/**
* Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this
* helper (imported by many paid tests) does not pull the quarantine registry
* into their touchfile closure. test/judge-panel.test.ts pins the two equal.
*/
export const JUDGE_PANEL_SAMPLES = 3;
/**
* Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of
* independent samples of the SAME prompt concurrently, inside its unchanged
* JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
* against the unchanged minimum; boolean fields gate on a strict majority.
* A sample that errors (refusal, truncation, non-JSON, malformed field) fails
* the whole panel and is never resampled. callJudge's 429 backoff happens
* before any model output exists, so it is transport, not a verdict retry.
*/
export async function judgePanel<T>(sample: () => Promise<T>): Promise<T[]> {
const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample()));
const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []);
if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult<T>).value);
const first = failures[0]!;
// A refusal is an unscored panel only when EVERY sample refused; a partial
// refusal beside scored samples is an ordinary failed panel.
if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) {
throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`);
}
throw first.reason;
}
/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */
export function judgePanelMean<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, keys: readonly K[]): Record<K, number> {
if (samples.length === 0) throw new Error('Judge panel has no samples');
return Object.fromEntries(keys.map(key => {
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value));
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`);
return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length];
})) as Record<K, number>;
}
/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */
export function judgePanelMajority<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, key: K): boolean {
if (samples.length === 0) throw new Error('Judge panel has no samples');
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
const bad = values.findIndex(value => typeof value !== 'boolean');
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`);
return values.filter(value => value === true).length * 2 > values.length;
}
/** Sample reasoning lines, numbered, for the collector record. */
export function judgePanelReasoning(samples: ReadonlyArray<unknown>): string {
return samples.map((sample, index) => {
const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined;
return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`;
}).join('\n');
}
/**
* Score documentation quality on clarity/completeness/actionability (1-5).
*/
+64
View File
@@ -59,3 +59,67 @@ export const CASE_CI_EXCLUDE: Record<string, { reason: string; tracking: string
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
};
/**
* Paid-eval verdict policy, pre-registered (approved 2026-09-29). Frozen before
* the census: any change after seeing census results needs Garry's
* re-approval and a fresh census, and bumps `version` (every trial record
* carries it as policy_version, so pass-rate history segments at the change).
* panel - behavior cases and quarantined cases run n independent
* trials; a behavior panel PASSES at >= k passing trials with
* no contract violation. Rule and judge cases run one trial.
* quarantine - entry below `entry.rate` per trial over >= `entry.minTrials`
* new-policy trials; exit at >= `exit.rate` over >=
* `exit.minTrials`; at most `capFraction` of each tier's
* blocking cases; an entry expires after `expiryWeeklyRuns`.
* judge - a judge case draws `samples` independent samples of one
* prompt concurrently; numeric dimensions gate on the panel
* mean against the unchanged threshold, booleans on a strict
* majority; an erroring sample fails the panel, never resampled.
* drift - one-sided Fisher exact alarm between input-identity series
* (Holm-controlled across the cases tested in one report).
* infraRedispatch - a census whose every red verdict is machine-classified
* INFRA or INCOMPLETE may be re-dispatched this many times as
* a new run; both runs are reported.
*/
export const EVAL_POLICY = {
version: 1,
panel: { n: 3, k: 2 },
quarantine: {
entry: { rate: 0.95, minTrials: 10 },
exit: { rate: 0.97, minTrials: 10 },
capFraction: 0.10,
expiryWeeklyRuns: 8,
},
judge: { samples: 3 },
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
infraRedispatch: 1,
} as const;
/**
* Quarantined paid cases, keyed by registry id (an E2E_TIERS key). A
* quarantined case still runs its full panel and reports in every lane, but
* its failed verdict cannot fail the lane unless the panel is a hard break
* (0 of n) or a trial violated a contract; it never counts as passing
* coverage. An entry needs the entry rule met on the current input identity,
* a written diagnosis that the failures are detector, harness or model-latency
* failures (a product defect is never quarantined), and unchanged case
* touchfiles in the change that adds it. Pinned by
* test/periodic-exclude-policy.test.ts.
* reason - the written diagnosis, with the pass-rate evidence
* failureClass - what the diagnosis found; a product defect has no class here
* tracking - issue or TODOS pointer
* owner - who removes it
* enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here)
* exit - the measurable exit condition
* At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and
* periodic are the blocking tiers) may be quarantined at once.
*/
export const CASE_QUARANTINE: Record<string, {
reason: string;
failureClass: 'detector' | 'harness' | 'model-latency';
tracking: string;
owner: string;
enteredAt: string;
exit: string;
}> = {};
+16 -3
View File
@@ -34,6 +34,8 @@ import {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
} from './touchfiles-data';
/** Repo-relative path of the pure-data file (the map-diff subject). */
@@ -145,6 +147,9 @@ export interface TouchfileMaps {
E2E_TIERS: Record<string, string>;
LLM_JUDGE_TOUCHFILES: Record<string, string[]>;
GLOBAL_TOUCHFILES: string[];
/** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */
E2E_KINDS?: Record<string, string>;
BEHAVIOR_WHY?: Record<string, string>;
}
export type MapDiffCause =
@@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
};
function isStringArray(v: unknown): v is string[] {
@@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps {
return isRecordOfStringArrays(o.E2E_TOUCHFILES)
&& isRecordOfStrings(o.E2E_TIERS)
&& isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES)
&& isStringArray(o.GLOBAL_TOUCHFILES);
&& isStringArray(o.GLOBAL_TOUCHFILES)
&& (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS))
&& (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY));
}
/**
* Pure map-diff core (injectable for tests — no git, no filesystem).
*
* A key counts as CHANGED when it was added to any per-key map, its dep-list
* array differs, or its tier value flipped. A key counts as REMOVED only when
* array differs, or its tier, kind or behavior tolerance changed. A per-key
* map missing on the old side (a base revision older than E2E_KINDS /
* BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when
* it is gone from every new per-key map; a key dropped from one map but still
* present in another (e.g. tier entry deleted, touchfile entry kept) counts
* as changed — conservative, because the test still exists with a different
@@ -212,7 +223,7 @@ export function diffTouchfileMapsCore(
oldMaps: TouchfileMaps,
newMaps: TouchfileMaps,
): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } {
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const;
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const;
const changed = new Set<string>();
const rawRemoved = new Set<string>();
@@ -290,6 +301,8 @@ export function diffTouchfileMaps(
' E2E_TIERS: m.E2E_TIERS,',
' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,',
' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,',
' E2E_KINDS: m.E2E_KINDS,',
' BEHAVIOR_WHY: m.BEHAVIOR_WHY,',
'}));',
'',
].join('\n'));
+303
View File
@@ -1573,3 +1573,306 @@ export const GLOBAL_TOUCHFILES = [
// diffed per key, so a data-only edit runs just the affected tests.
// Map-diff fails CLOSED — any error on that path still runs everything.
];
/**
* Eval kind per live case (every E2E_TIERS and LLM_JUDGE_TOUCHFILES key).
* The kind fixes the trial policy before the run (EVAL_POLICY in
* periodic-exclude-data.ts):
* rule - one trial; any failed assertion fails the verdict. The default.
* behavior - a panel of independent trials, PASS at the policy majority;
* needs a BEHAVIOR_WHY entry naming the tolerated deviation.
* judge - an LLM-judge score of a static input, sampled as a panel.
* Reclassification is a reviewed diff, never a runtime switch.
*/
export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'ship-skipped-queued-finding': 'rule',
'investigate-owned-completion': 'rule',
'investigate-owned-abort': 'rule',
'investigate-owned-ending-error': 'rule',
'shared-libs-review-path-eligibility': 'rule',
'shared-libs-review-index-flags': 'rule',
'shared-libs-review-prior-coverage': 'rule',
'shared-libs-codex-read-only': 'rule',
'shared-libs-read-only': 'rule',
'shared-libs-unsupported-git': 'rule',
'shared-libs-review-lifecycle': 'rule',
'shared-libs-review-revalidation': 'rule',
'shared-libs-opportunity-judgment': 'behavior',
'shared-libs-pr-coverage': 'rule',
'shared-libs-plan-callers': 'rule',
'browse-basic': 'rule',
'browse-snapshot': 'rule',
'aside-browse-basic': 'rule',
'aside-browse-flow': 'rule',
'aside-qa-quick': 'rule',
'aside-scrape-json': 'rule',
'aside-canary-quick': 'rule',
'hermetic-canary': 'rule',
'hermetic-sentinel': 'rule',
'skillmd-setup-discovery': 'rule',
'skillmd-no-local-binary': 'rule',
'skillmd-outside-git': 'rule',
'session-awareness': 'rule',
'operational-learning': 'rule',
'first-task-scaffold': 'rule',
'qa-quick': 'rule',
'qa-b6-static': 'rule',
'qa-b7-spa': 'rule',
'qa-b8-checkout': 'rule',
'qa-only-no-fix': 'rule',
'qa-fix-loop': 'rule',
'qa-bootstrap': 'rule',
'review-exploratory-small-cli': 'rule',
'ship-exploratory-small-cli': 'rule',
'ship-exploratory-unavailable': 'rule',
'ship-exploratory-plan-checks': 'rule',
'ship-exploratory-late-input': 'rule',
'qa-functional-cli-report': 'rule',
'qa-functional-webhook-report': 'rule',
'qa-functional-cli-fix': 'rule',
'qa-functional-webhook-fix': 'rule',
'review-sql-injection': 'rule',
'review-enum-completeness': 'rule',
'review-base-branch': 'rule',
'review-design-lite': 'behavior',
'review-coverage-audit': 'rule',
'review-dashboard-via': 'rule',
'review-army-migration-safety': 'rule',
'review-army-perf-n-plus-one': 'rule',
'review-army-delivery-audit': 'rule',
'review-army-quality-score': 'rule',
'review-army-json-findings': 'rule',
'review-army-red-team': 'behavior',
'review-army-consensus': 'behavior',
'review-army-simplification': 'behavior',
'review-army-simplification-precision': 'behavior',
'office-hours-spec-review': 'rule',
'office-hours-brain-writeback': 'behavior',
'gbrain-roundtrip-local': 'rule',
'sync-gbrain-read-ready': 'rule',
'sync-gbrain-read-unknown': 'rule',
'office-hours-forcing-energy': 'behavior',
'office-hours-builder-wildness': 'behavior',
'plan-ceo-review': 'rule',
'plan-ceo-review-selective': 'rule',
'plan-ceo-review-benefits': 'rule',
'plan-ceo-review-expansion-energy': 'behavior',
'plan-eng-review': 'rule',
'plan-eng-review-artifact': 'rule',
'plan-eng-coverage-audit': 'rule',
'plan-review-report': 'rule',
'plan-ceo-review-plan-mode': 'rule',
'plan-eng-review-plan-mode': 'rule',
'plan-design-review-plan-mode': 'rule',
'plan-devex-review-plan-mode': 'rule',
'plan-mode-no-op': 'rule',
'office-hours-auto-mode': 'rule',
'auto-decide-preserved': 'rule',
'auq-format-gate': 'rule',
'plan-ceo-mode-routing': 'rule',
'plan-design-with-ui-scope': 'rule',
'tpa-present': 'rule',
'tpa-absent-linux': 'rule',
'tpa-broken': 'rule',
'tpa-absent-darwin': 'rule',
'tpa-apple-ban': 'rule',
'ship-section-loading': 'rule',
'plan-ceo-section-loading': 'rule',
'carve-section-loading': 'rule',
'plan-eng-finding-floor': 'rule',
'plan-ceo-finding-floor': 'rule',
'plan-design-finding-floor': 'rule',
'plan-devex-finding-floor': 'rule',
'plan-eng-multi-finding-batching': 'rule',
'plan-ceo-split-overflow': 'rule',
'setup-gbrain-remote': 'rule',
'setup-gbrain-bad-token': 'rule',
'setup-gbrain-path4-local-pglite': 'rule',
'plan-ceo-review-format-mode': 'behavior',
'plan-ceo-review-format-approach': 'behavior',
'plan-eng-review-format-coverage': 'behavior',
'plan-eng-review-format-kind': 'behavior',
'office-hours-phase4-fork': 'behavior',
'llm-judge-recommendation': 'judge',
'plan-ceo-review-prosons-cadence': 'behavior',
'plan-review-prosons-format': 'behavior',
'plan-review-prosons-hardstop-neg': 'behavior',
'plan-review-prosons-neutral-neg': 'behavior',
'plan-tune-inspect': 'rule',
'codex-offered-office-hours': 'rule',
'codex-offered-ceo-review': 'rule',
'codex-offered-design-review': 'rule',
'codex-offered-eng-review': 'rule',
'timeline-event-flow': 'rule',
'context-recovery-artifacts': 'rule',
'context-save-writes-file': 'rule',
'context-restore-loads-latest': 'rule',
'context-save-routing': 'rule',
'context-save-then-restore-roundtrip': 'rule',
'context-restore-fragment-match': 'rule',
'context-restore-empty-state': 'rule',
'context-restore-list-delegates': 'rule',
'context-restore-legacy-compat': 'rule',
'context-save-list-current-branch': 'rule',
'context-save-list-all-branches': 'rule',
'ship-base-branch': 'rule',
'ship-local-workflow': 'rule',
'ship-managed-hook-refresh': 'rule',
'ship-unmanaged-hook-consent': 'rule',
'ship-local-hook-preservation': 'rule',
'ship-coverage-audit': 'rule',
'ship-triage': 'rule',
'ship-docsync-missing-marker': 'rule',
'ship-docsync-missing-asset': 'rule',
'ship-docsync-launch-failure': 'rule',
'ship-docsync-timeout-unsettled': 'rule',
'ship-docsync-late-result': 'rule',
'ship-docsync-stale-before': 'rule',
'ship-docsync-stale-after': 'rule',
'ship-docsync-recovery': 'rule',
'ship-docsync-completion': 'rule',
'ship-docsync-current': 'rule',
'ship-docsync-failure': 'rule',
'ship-docsync-store': 'rule',
'docsync-spawned': 'rule',
'retro': 'rule',
'retro-base-branch': 'rule',
'cso-full-audit': 'rule',
'cso-diff-mode': 'rule',
'cso-infra-scope': 'rule',
'learnings-show': 'rule',
'document-release': 'rule',
'codex-review': 'rule',
'codex-discover-skill': 'rule',
'codex-review-findings': 'rule',
'outside-voice-codex-to-claude-code': 'rule',
'outside-voice-claude-code-to-codex': 'rule',
'outside-plan-disabled-no-fallback': 'rule',
'codex-sol-scope-termination': 'rule',
'design-consultation-core': 'rule',
'design-consultation-existing': 'rule',
'design-consultation-research': 'rule',
'design-consultation-preview': 'rule',
'plan-design-review-no-ui-scope': 'rule',
'design-review-fix': 'rule',
'design-review-detector-shim': 'rule',
'design-review-detector-shim-dom': 'rule',
'design-review-plugin-handoff': 'rule',
'design-html-slop-gate': 'behavior',
'diagram-triplet': 'rule',
'diagram-authoring-quality': 'rule',
'gstack-upgrade-happy-path': 'rule',
'land-and-deploy-workflow': 'rule',
'land-and-deploy-first-run': 'rule',
'land-and-deploy-review-gate': 'rule',
'canary-workflow': 'rule',
'benchmark-workflow': 'rule',
'setup-deploy-workflow': 'rule',
'autoplan-dual-voice': 'rule',
'benchmark-providers-live': 'rule',
'scrape-match-path': 'behavior',
'scrape-prototype-path': 'behavior',
'skillify-happy-path': 'rule',
'skillify-provenance-refusal': 'rule',
'skillify-approval-reject': 'rule',
'journey-ideation': 'rule',
'journey-plan-eng': 'rule',
'journey-debug': 'rule',
'journey-qa': 'rule',
'journey-code-review': 'rule',
'journey-ship': 'rule',
'journey-docs': 'rule',
'journey-retro': 'rule',
'journey-design-system': 'rule',
'journey-visual-qa': 'rule',
'ios-qa-device': 'rule',
'arm-benchmark-native-overbuild': 'rule',
'arm-benchmark-crud-endpoint': 'rule',
'arm-benchmark-bugfix-decoys': 'rule',
'office-hours-section-loading': 'rule',
'office-hours-design-draft': 'rule',
'plan-decision-classification': 'rule',
'plan-devex-peer-comparison-classification': 'rule',
'health-reporting': 'rule',
'overlay-harness-claude-dedicated-tools-vs-bash': 'rule',
'overlay-harness-opus-4-7-effort-match-trivial': 'rule',
'overlay-harness-opus-4-7-literal-interpretation': 'rule',
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule',
'journey-negatives': 'rule',
'review/SKILL.md workflow': 'judge',
'setup-browser-cookies/SKILL.md workflow': 'judge',
'browse/SKILL.md reference': 'judge',
'setup block': 'judge',
'qa/SKILL.md workflow': 'judge',
'qa/SKILL.md health rubric': 'judge',
'qa/SKILL.md anti-refusal': 'judge',
'cross-skill greptile consistency': 'judge',
'ship/SKILL.md workflow': 'judge',
'document-release/SKILL.md workflow': 'judge',
'plan-ceo-review/SKILL.md modes': 'judge',
'plan-eng-review/SKILL.md sections': 'judge',
'plan-design-review/SKILL.md passes': 'judge',
'design-review/SKILL.md fix loop': 'judge',
'design-consultation/SKILL.md research': 'judge',
'land-and-deploy/SKILL.md workflow': 'judge',
'canary/SKILL.md monitoring loop': 'judge',
'benchmark/SKILL.md perf collection': 'judge',
'setup-deploy/SKILL.md platform setup': 'judge',
'retro/SKILL.md instructions': 'judge',
'qa-only/SKILL.md workflow': 'judge',
'gstack-upgrade/SKILL.md upgrade flow': 'judge',
'sync-gbrain/SKILL.md read-only readiness': 'judge',
'voice directive tone': 'judge',
};
/**
* One-line tolerance for every behavior-kind case: why an occasional
* deviation is acceptable product behavior. Keys equal the behavior ids of
* E2E_KINDS; values are non-empty.
*/
export const BEHAVIOR_WHY: Record<string, string> = {
'shared-libs-opportunity-judgment':
"Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.",
'review-design-lite':
"How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.",
'review-army-red-team':
"Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.",
'review-army-consensus':
"Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.",
'review-army-simplification':
"Flagging the planted unnecessary structure is an advisory-lens quality judgment.",
'review-army-simplification-precision':
"Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.",
'office-hours-forcing-energy':
"The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.",
'office-hours-builder-wildness':
"Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.",
'office-hours-brain-writeback':
"The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.",
'office-hours-phase4-fork':
"Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.",
'plan-ceo-review-expansion-energy':
"Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.",
'plan-ceo-review-format-mode':
"Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.",
'plan-ceo-review-format-approach':
"Approach-menu Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-coverage':
"Coverage-issue Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-kind':
"Kind-note wording is live formatting of one AskUserQuestion.",
'plan-ceo-review-prosons-cadence':
"Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.",
'plan-review-prosons-format':
"The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.",
'plan-review-prosons-hardstop-neg':
"Not using the hard-stop escape on an ordinary decision is live formatting of one question.",
'plan-review-prosons-neutral-neg':
"Avoiding neutral posture and naming a because-reason is live formatting of one question.",
'design-html-slop-gate':
"How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.",
'scrape-match-path':
"The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.",
'scrape-prototype-path':
"The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.",
};
+24 -11
View File
@@ -4,7 +4,7 @@ import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
import { JUDGE_MS } from './eval-budgets';
import type { JudgeScore } from './llm-judge';
import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input';
import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache,
type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache';
@@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres
|| typeof value.reasoning !== 'string'
|| (structuredResponse && (!value.reasoning.trim()
|| value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false;
return (['clarity', 'completeness', 'actionability'] as const).every(key =>
return JUDGE_SCORE_DIMENSIONS.every(key =>
typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5);
}
const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 };
/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */
export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array<JudgeScore & EvalCacheValue> } {
if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples'
|| !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES
|| !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false;
const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS);
return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]);
}
export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null;
lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null;
/** The attempt guard is rechecked after synchronous input/provenance reads. */
publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined;
publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined;
} {
const env = opts.env ?? process.env;
const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined };
const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined };
const pr = Number(env.EVALS_CACHE_PR);
// Runtime ID is the immutable CI image manifest, not a mutable image tag.
// Nonstandard Node/Bun preload code or custom model endpoints need a separate
@@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
prompts: { [opts.testName]: prompt },
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1,
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0,
panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' },
...(opts.stream ? { stream: true } : {}),
...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) },
@@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
return {
lookup() {
const result = lookupEvalInputCache({ ...common, identity: before,
validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) });
validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) });
return result.status === 'reused'
? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null;
? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null;
},
publish(scores, isActive = () => true) {
publish(samples, isActive = () => true) {
// Caller reaches here ONLY after its actual assertions passed. A later
// failed case in the file does not erase this independently completed case.
if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) };
if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
const after = currentIdentity();
const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID;
if (!after || !runId || !isActive()) return;
@@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
cancelled: false, skipped: 0, failed: 0, passed: 1,
cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }],
source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() },
result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning },
result: panel,
} });
// A slow synchronous write can consume the recording allowance. The
// caller withdraws this new receipt if its final deadline check fails.