Merge remote-tracking branch 'origin/capy/rel-b' into capy/rel-a

This commit is contained in:
garrytan committed 2026-09-29 19:24:03 +00:00
commit a385e5de18
16 files changed
+1632 -234

No files matched your search

+59
View File
@@ -23,6 +23,8 @@ export interface JudgeScore {
reasoning: string;
}
export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const;
export interface JudgeRefusalEvidence {
stop_reason: 'refusal';
response_id: string | null;
@@ -196,6 +198,63 @@ export async function callJudge<T>(
}
}
/**
* Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this
* helper (imported by many paid tests) does not pull the quarantine registry
* into their touchfile closure. test/judge-panel.test.ts pins the two equal.
*/
export const JUDGE_PANEL_SAMPLES = 3;
/**
* Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of
* independent samples of the SAME prompt concurrently, inside its unchanged
* JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
* against the unchanged minimum; boolean fields gate on a strict majority.
* A sample that errors (refusal, truncation, non-JSON, malformed field) fails
* the whole panel and is never resampled. callJudge's 429 backoff happens
* before any model output exists, so it is transport, not a verdict retry.
*/
export async function judgePanel<T>(sample: () => Promise<T>): Promise<T[]> {
const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample()));
const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []);
if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult<T>).value);
const first = failures[0]!;
// A refusal is an unscored panel only when EVERY sample refused; a partial
// refusal beside scored samples is an ordinary failed panel.
if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) {
throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`);
}
throw first.reason;
}
/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */
export function judgePanelMean<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, keys: readonly K[]): Record<K, number> {
if (samples.length === 0) throw new Error('Judge panel has no samples');
return Object.fromEntries(keys.map(key => {
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value));
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`);
return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length];
})) as Record<K, number>;
}
/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */
export function judgePanelMajority<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, key: K): boolean {
if (samples.length === 0) throw new Error('Judge panel has no samples');
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
const bad = values.findIndex(value => typeof value !== 'boolean');
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`);
return values.filter(value => value === true).length * 2 > values.length;
}
/** Sample reasoning lines, numbered, for the collector record. */
export function judgePanelReasoning(samples: ReadonlyArray<unknown>): string {
return samples.map((sample, index) => {
const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined;
return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`;
}).join('\n');
}
/**
* Score documentation quality on clarity/completeness/actionability (1-5).
*/
+16 -6
View File
@@ -72,7 +72,12 @@ export const CASE_CI_EXCLUDE: Record<string, { reason: string; tracking: string
* new-policy trials; exit at >= `exit.rate` over >=
* `exit.minTrials`; at most `capFraction` of each tier's
* blocking cases; an entry expires after `expiryWeeklyRuns`.
* drift - one-sided Fisher exact alarm between input-identity series.
* judge - a judge case draws `samples` independent samples of one
* prompt concurrently; numeric dimensions gate on the panel
* mean against the unchanged threshold, booleans on a strict
* majority; an erroring sample fails the panel, never resampled.
* drift - one-sided Fisher exact alarm between input-identity series
* (Holm-controlled across the cases tested in one report).
* infraRedispatch - a census whose every red verdict is machine-classified
* INFRA or INCOMPLETE may be re-dispatched this many times as
* a new run; both runs are reported.
@@ -86,6 +91,7 @@ export const EVAL_POLICY = {
capFraction: 0.10,
expiryWeeklyRuns: 8,
},
judge: { samples: 3 },
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
infraRedispatch: 1,
} as const;
@@ -100,14 +106,18 @@ export const EVAL_POLICY = {
* failures (a product defect is never quarantined), and unchanged case
* touchfiles in the change that adds it. Pinned by
* test/periodic-exclude-policy.test.ts.
* reason - the written diagnosis
* tracking - issue or TODOS pointer
* owner - who removes it
* enteredAt - ISO date the entry landed (expiry counts weekly runs from here)
* exit - the measurable exit condition
* reason - the written diagnosis, with the pass-rate evidence
* failureClass - what the diagnosis found; a product defect has no class here
* tracking - issue or TODOS pointer
* owner - who removes it
* enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here)
* exit - the measurable exit condition
* At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and
* periodic are the blocking tiers) may be quarantined at once.
*/
export const CASE_QUARANTINE: Record<string, {
reason: string;
failureClass: 'detector' | 'harness' | 'model-latency';
tracking: string;
owner: string;
enteredAt: string;
+16 -3
View File
@@ -34,6 +34,8 @@ import {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
} from './touchfiles-data';
/** Repo-relative path of the pure-data file (the map-diff subject). */
@@ -145,6 +147,9 @@ export interface TouchfileMaps {
E2E_TIERS: Record<string, string>;
LLM_JUDGE_TOUCHFILES: Record<string, string[]>;
GLOBAL_TOUCHFILES: string[];
/** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */
E2E_KINDS?: Record<string, string>;
BEHAVIOR_WHY?: Record<string, string>;
}
export type MapDiffCause =
@@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
};
function isStringArray(v: unknown): v is string[] {
@@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps {
return isRecordOfStringArrays(o.E2E_TOUCHFILES)
&& isRecordOfStrings(o.E2E_TIERS)
&& isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES)
&& isStringArray(o.GLOBAL_TOUCHFILES);
&& isStringArray(o.GLOBAL_TOUCHFILES)
&& (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS))
&& (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY));
}
/**
* Pure map-diff core (injectable for tests — no git, no filesystem).
*
* A key counts as CHANGED when it was added to any per-key map, its dep-list
* array differs, or its tier value flipped. A key counts as REMOVED only when
* array differs, or its tier, kind or behavior tolerance changed. A per-key
* map missing on the old side (a base revision older than E2E_KINDS /
* BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when
* it is gone from every new per-key map; a key dropped from one map but still
* present in another (e.g. tier entry deleted, touchfile entry kept) counts
* as changed — conservative, because the test still exists with a different
@@ -212,7 +223,7 @@ export function diffTouchfileMapsCore(
oldMaps: TouchfileMaps,
newMaps: TouchfileMaps,
): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } {
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const;
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const;
const changed = new Set<string>();
const rawRemoved = new Set<string>();
@@ -290,6 +301,8 @@ export function diffTouchfileMaps(
' E2E_TIERS: m.E2E_TIERS,',
' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,',
' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,',
' E2E_KINDS: m.E2E_KINDS,',
' BEHAVIOR_WHY: m.BEHAVIOR_WHY,',
'}));',
'',
].join('\n'));
+93 -48
View File
@@ -1597,7 +1597,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'shared-libs-unsupported-git': 'rule',
'shared-libs-review-lifecycle': 'rule',
'shared-libs-review-revalidation': 'rule',
'shared-libs-opportunity-judgment': 'rule',
'shared-libs-opportunity-judgment': 'behavior',
'shared-libs-pr-coverage': 'rule',
'shared-libs-plan-callers': 'rule',
'browse-basic': 'rule',
@@ -1634,7 +1634,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'review-sql-injection': 'rule',
'review-enum-completeness': 'rule',
'review-base-branch': 'rule',
'review-design-lite': 'rule',
'review-design-lite': 'behavior',
'review-coverage-audit': 'rule',
'review-dashboard-via': 'rule',
'review-army-migration-safety': 'rule',
@@ -1642,21 +1642,21 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'review-army-delivery-audit': 'rule',
'review-army-quality-score': 'rule',
'review-army-json-findings': 'rule',
'review-army-red-team': 'rule',
'review-army-consensus': 'rule',
'review-army-simplification': 'rule',
'review-army-simplification-precision': 'rule',
'review-army-red-team': 'behavior',
'review-army-consensus': 'behavior',
'review-army-simplification': 'behavior',
'review-army-simplification-precision': 'behavior',
'office-hours-spec-review': 'rule',
'office-hours-brain-writeback': 'rule',
'office-hours-brain-writeback': 'behavior',
'gbrain-roundtrip-local': 'rule',
'sync-gbrain-read-ready': 'rule',
'sync-gbrain-read-unknown': 'rule',
'office-hours-forcing-energy': 'rule',
'office-hours-builder-wildness': 'rule',
'office-hours-forcing-energy': 'behavior',
'office-hours-builder-wildness': 'behavior',
'plan-ceo-review': 'rule',
'plan-ceo-review-selective': 'rule',
'plan-ceo-review-benefits': 'rule',
'plan-ceo-review-expansion-energy': 'rule',
'plan-ceo-review-expansion-energy': 'behavior',
'plan-eng-review': 'rule',
'plan-eng-review-artifact': 'rule',
'plan-eng-coverage-audit': 'rule',
@@ -1688,16 +1688,16 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'setup-gbrain-remote': 'rule',
'setup-gbrain-bad-token': 'rule',
'setup-gbrain-path4-local-pglite': 'rule',
'plan-ceo-review-format-mode': 'rule',
'plan-ceo-review-format-approach': 'rule',
'plan-eng-review-format-coverage': 'rule',
'plan-eng-review-format-kind': 'rule',
'office-hours-phase4-fork': 'rule',
'llm-judge-recommendation': 'rule',
'plan-ceo-review-prosons-cadence': 'rule',
'plan-review-prosons-format': 'rule',
'plan-review-prosons-hardstop-neg': 'rule',
'plan-review-prosons-neutral-neg': 'rule',
'plan-ceo-review-format-mode': 'behavior',
'plan-ceo-review-format-approach': 'behavior',
'plan-eng-review-format-coverage': 'behavior',
'plan-eng-review-format-kind': 'behavior',
'office-hours-phase4-fork': 'behavior',
'llm-judge-recommendation': 'judge',
'plan-ceo-review-prosons-cadence': 'behavior',
'plan-review-prosons-format': 'behavior',
'plan-review-prosons-hardstop-neg': 'behavior',
'plan-review-prosons-neutral-neg': 'behavior',
'plan-tune-inspect': 'rule',
'codex-offered-office-hours': 'rule',
'codex-offered-ceo-review': 'rule',
@@ -1758,7 +1758,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'design-review-detector-shim': 'rule',
'design-review-detector-shim-dom': 'rule',
'design-review-plugin-handoff': 'rule',
'design-html-slop-gate': 'rule',
'design-html-slop-gate': 'behavior',
'diagram-triplet': 'rule',
'diagram-authoring-quality': 'rule',
'gstack-upgrade-happy-path': 'rule',
@@ -1770,8 +1770,8 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'setup-deploy-workflow': 'rule',
'autoplan-dual-voice': 'rule',
'benchmark-providers-live': 'rule',
'scrape-match-path': 'rule',
'scrape-prototype-path': 'rule',
'scrape-match-path': 'behavior',
'scrape-prototype-path': 'behavior',
'skillify-happy-path': 'rule',
'skillify-provenance-refusal': 'rule',
'skillify-approval-reject': 'rule',
@@ -1799,30 +1799,30 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'overlay-harness-opus-4-7-literal-interpretation': 'rule',
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule',
'journey-negatives': 'rule',
'review/SKILL.md workflow': 'rule',
'setup-browser-cookies/SKILL.md workflow': 'rule',
'browse/SKILL.md reference': 'rule',
'setup block': 'rule',
'qa/SKILL.md workflow': 'rule',
'qa/SKILL.md health rubric': 'rule',
'qa/SKILL.md anti-refusal': 'rule',
'cross-skill greptile consistency': 'rule',
'ship/SKILL.md workflow': 'rule',
'document-release/SKILL.md workflow': 'rule',
'plan-ceo-review/SKILL.md modes': 'rule',
'plan-eng-review/SKILL.md sections': 'rule',
'plan-design-review/SKILL.md passes': 'rule',
'design-review/SKILL.md fix loop': 'rule',
'design-consultation/SKILL.md research': 'rule',
'land-and-deploy/SKILL.md workflow': 'rule',
'canary/SKILL.md monitoring loop': 'rule',
'benchmark/SKILL.md perf collection': 'rule',
'setup-deploy/SKILL.md platform setup': 'rule',
'retro/SKILL.md instructions': 'rule',
'qa-only/SKILL.md workflow': 'rule',
'gstack-upgrade/SKILL.md upgrade flow': 'rule',
'sync-gbrain/SKILL.md read-only readiness': 'rule',
'voice directive tone': 'rule',
'review/SKILL.md workflow': 'judge',
'setup-browser-cookies/SKILL.md workflow': 'judge',
'browse/SKILL.md reference': 'judge',
'setup block': 'judge',
'qa/SKILL.md workflow': 'judge',
'qa/SKILL.md health rubric': 'judge',
'qa/SKILL.md anti-refusal': 'judge',
'cross-skill greptile consistency': 'judge',
'ship/SKILL.md workflow': 'judge',
'document-release/SKILL.md workflow': 'judge',
'plan-ceo-review/SKILL.md modes': 'judge',
'plan-eng-review/SKILL.md sections': 'judge',
'plan-design-review/SKILL.md passes': 'judge',
'design-review/SKILL.md fix loop': 'judge',
'design-consultation/SKILL.md research': 'judge',
'land-and-deploy/SKILL.md workflow': 'judge',
'canary/SKILL.md monitoring loop': 'judge',
'benchmark/SKILL.md perf collection': 'judge',
'setup-deploy/SKILL.md platform setup': 'judge',
'retro/SKILL.md instructions': 'judge',
'qa-only/SKILL.md workflow': 'judge',
'gstack-upgrade/SKILL.md upgrade flow': 'judge',
'sync-gbrain/SKILL.md read-only readiness': 'judge',
'voice directive tone': 'judge',
};
/**
@@ -1830,4 +1830,49 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
* deviation is acceptable product behavior. Keys equal the behavior ids of
* E2E_KINDS; values are non-empty.
*/
export const BEHAVIOR_WHY: Record<string, string> = {};
export const BEHAVIOR_WHY: Record<string, string> = {
'shared-libs-opportunity-judgment':
"Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.",
'review-design-lite':
"How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.",
'review-army-red-team':
"Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.",
'review-army-consensus':
"Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.",
'review-army-simplification':
"Flagging the planted unnecessary structure is an advisory-lens quality judgment.",
'review-army-simplification-precision':
"Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.",
'office-hours-forcing-energy':
"The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.",
'office-hours-builder-wildness':
"Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.",
'office-hours-brain-writeback':
"The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.",
'office-hours-phase4-fork':
"Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.",
'plan-ceo-review-expansion-energy':
"Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.",
'plan-ceo-review-format-mode':
"Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.",
'plan-ceo-review-format-approach':
"Approach-menu Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-coverage':
"Coverage-issue Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-kind':
"Kind-note wording is live formatting of one AskUserQuestion.",
'plan-ceo-review-prosons-cadence':
"Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.",
'plan-review-prosons-format':
"The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.",
'plan-review-prosons-hardstop-neg':
"Not using the hard-stop escape on an ordinary decision is live formatting of one question.",
'plan-review-prosons-neutral-neg':
"Avoiding neutral posture and naming a because-reason is live formatting of one question.",
'design-html-slop-gate':
"How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.",
'scrape-match-path':
"The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.",
'scrape-prototype-path':
"The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.",
};
+24 -11
View File
@@ -4,7 +4,7 @@ import * as path from 'node:path';
import { spawnSync } from 'node:child_process';
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
import { JUDGE_MS } from './eval-budgets';
import type { JudgeScore } from './llm-judge';
import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input';
import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache,
type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache';
@@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres
|| typeof value.reasoning !== 'string'
|| (structuredResponse && (!value.reasoning.trim()
|| value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false;
return (['clarity', 'completeness', 'actionability'] as const).every(key =>
return JUDGE_SCORE_DIMENSIONS.every(key =>
typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5);
}
const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 };
/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */
export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array<JudgeScore & EvalCacheValue> } {
if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples'
|| !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES
|| !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false;
const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS);
return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]);
}
export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null;
lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null;
/** The attempt guard is rechecked after synchronous input/provenance reads. */
publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined;
publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined;
} {
const env = opts.env ?? process.env;
const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined };
const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined };
const pr = Number(env.EVALS_CACHE_PR);
// Runtime ID is the immutable CI image manifest, not a mutable image tag.
// Nonstandard Node/Bun preload code or custom model endpoints need a separate
@@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
prompts: { [opts.testName]: prompt },
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1,
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0,
panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' },
...(opts.stream ? { stream: true } : {}),
...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) },
@@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
return {
lookup() {
const result = lookupEvalInputCache({ ...common, identity: before,
validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) });
validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) });
return result.status === 'reused'
? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null;
? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null;
},
publish(scores, isActive = () => true) {
publish(samples, isActive = () => true) {
// Caller reaches here ONLY after its actual assertions passed. A later
// failed case in the file does not erase this independently completed case.
if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) };
if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
const after = currentIdentity();
const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID;
if (!after || !runId || !isActive()) return;
@@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
cancelled: false, skipped: 0, failed: 0, passed: 1,
cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }],
source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() },
result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning },
result: panel,
} });
// A slow synchronous write can consume the recording allowance. The
// caller withdraws this new receipt if its final deadline check fails.