mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
Merge remote-tracking branch 'origin/capy/rel-a' into capy/rel-c
This commit is contained in:
commit
b2ca207cf0
46 files changed
+4936
-936
No files matched your search
+48
-100
@@ -47,132 +47,80 @@ export const ALL_TIERS = {
|
||||
export const SHARD_RESERVE_MS = 2 * 60_000;
|
||||
|
||||
/**
|
||||
* Retry policy (approved 2026-09-29): a timed-out attempt is a verdict. Bun's
|
||||
* --retry reruns a failed case after it may have spent its whole budget, so an
|
||||
* automatic retry is kept only where one more attempt is short: every case of
|
||||
* the file has a per-attempt budget of at most RETRY_MAX_CASE_MS, the CAPTURE
|
||||
* tier plus its recording grace. Those failures are fast flake classes (API
|
||||
* blips, tool hiccups) and a retry costs at most one more short attempt. Files
|
||||
* with any longer case run once. Per-case budgets never change with this rule.
|
||||
* Retry policy (approved 2026-09-29, eval reliability wave): paid evals never
|
||||
* retry. Each case's kind (E2E_KINDS) fixes its trials before the run: `rule`
|
||||
* one trial, `behavior` a panel of EVAL_POLICY.panel independent trials, and
|
||||
* `judge` one case that samples its judge panel internally. A failed verdict
|
||||
* is final for that run; a manual re-run adds trials under a new run attempt
|
||||
* and never replaces the original verdict. Rows below keep only wall
|
||||
* supervision; per-case budgets never change with this rule.
|
||||
*/
|
||||
export const RETRY_MAX_CASE_MS = CAPTURE_MS + 15_000;
|
||||
|
||||
export function retriesWithinCaseCap(caseMs: number, configuredRetries: number): number {
|
||||
return caseMs <= RETRY_MAX_CASE_MS ? configuredRetries : 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Unregistered paid files that keep one automatic retry: every case budget is
|
||||
* JUDGE or CAPTURE tier (test/paid-retry-supervision.test.ts scans each source).
|
||||
* Registered rows below derive retries from their declared caseMs; every other
|
||||
* paid file runs once.
|
||||
*/
|
||||
export const SHORT_CASE_RETRY_FILES: readonly string[] = [
|
||||
'test/codex-e2e-sol-scope.test.ts',
|
||||
'test/llm-judge-recommendation.test.ts',
|
||||
'test/skill-e2e-ask-user-question-format-compliance.test.ts',
|
||||
'test/skill-e2e-benchmark-providers.test.ts',
|
||||
'test/skill-e2e-bws.test.ts',
|
||||
'test/skill-e2e-context-skills.test.ts',
|
||||
'test/skill-e2e-coverage-audit.test.ts',
|
||||
'test/skill-e2e-diagram.test.ts',
|
||||
'test/skill-e2e-first-task-scaffold.test.ts',
|
||||
'test/skill-e2e-gbrain-roundtrip-local.test.ts',
|
||||
'test/skill-e2e-hermetic-canary.test.ts',
|
||||
'test/skill-e2e-investigate-owned-completion.test.ts',
|
||||
'test/skill-e2e-investigate-owned-termination.test.ts',
|
||||
'test/skill-e2e-learnings.test.ts',
|
||||
'test/skill-e2e-plan-tune.test.ts',
|
||||
'test/skill-e2e-qa-functional-fix.test.ts',
|
||||
'test/skill-e2e-qa-functional.test.ts',
|
||||
'test/skill-e2e-review-army.test.ts',
|
||||
'test/skill-e2e-review.test.ts',
|
||||
'test/skill-e2e-session-intelligence.test.ts',
|
||||
'test/skill-e2e-setup-gbrain-bad-token.test.ts',
|
||||
'test/skill-e2e-setup-gbrain-path4-local-pglite.test.ts',
|
||||
'test/skill-e2e-setup-gbrain-remote.test.ts',
|
||||
'test/skill-e2e-ship-hook-consent.test.ts',
|
||||
'test/skill-e2e-ship-hook-refresh.test.ts',
|
||||
'test/skill-e2e-ship-skip.test.ts',
|
||||
'test/skill-e2e-sync-gbrain-readiness.test.ts',
|
||||
'test/skill-e2e-third-party-actions.test.ts',
|
||||
'test/skill-e2e-triage.test.ts',
|
||||
'test/skill-routing-e2e.test.ts',
|
||||
];
|
||||
|
||||
/** Whole-file supervision covers every attempt the retry policy allows.
|
||||
* These fixtures allow 25 minutes per case, so they run once.
|
||||
/** Whole-file supervision for one run of every case.
|
||||
* These fixtures allow 25 minutes per case.
|
||||
* Reserve the sequential upper bound even when Bun runs sibling cases together.
|
||||
*/
|
||||
export const FINDING_RETRY_BUDGETS = [
|
||||
{ file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 },
|
||||
].map(({ file, cases }) => {
|
||||
const retries = retriesWithinCaseCap(1_500_000, 1);
|
||||
return {
|
||||
file, cases,
|
||||
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
testMs: 1_500_000,
|
||||
caseMs: 1_500_000,
|
||||
retries,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: cases * 1_500_000 * (retries + 1) + SHARD_RESERVE_MS,
|
||||
};
|
||||
});
|
||||
].map(({ file, cases }) => ({
|
||||
file, cases,
|
||||
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
testMs: 1_500_000,
|
||||
caseMs: 1_500_000,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: cases * 1_500_000 + SHARD_RESERVE_MS,
|
||||
}));
|
||||
|
||||
/** Three existing captures in one 16-minute case, so the file runs once. */
|
||||
/** Three existing captures in one 16-minute case. */
|
||||
export const AUQ_CONSISTENCY_RETRY_BUDGET = {
|
||||
file: 'test/skill-e2e-auq-consistency.test.ts',
|
||||
id: 'auq-consistency-existing-retry-v1',
|
||||
cases: 1,
|
||||
testMs: 3 * CAPTURE_MS + 60_000,
|
||||
caseMs: 3 * CAPTURE_MS + 60_000,
|
||||
retries: retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1),
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: (3 * CAPTURE_MS + 60_000) * (retriesWithinCaseCap(3 * CAPTURE_MS + 60_000, 1) + 1) + SHARD_RESERVE_MS,
|
||||
shardMs: 3 * CAPTURE_MS + 60_000 + SHARD_RESERVE_MS,
|
||||
} as const;
|
||||
|
||||
/** These fixtures have a fixed case count in every supported tier. */
|
||||
export const STRICT_RETRY_CASE_BUDGETS = [...FINDING_RETRY_BUDGETS, AUQ_CONSISTENCY_RETRY_BUDGET];
|
||||
|
||||
/** Whole-file walls cover all existing cases and every allowed attempt, even if
|
||||
* Bun runs them sequentially. Mixed-tier files reserve their larger complete
|
||||
* tier, never a currently selected subset. caseMs is the longest single case
|
||||
* budget, which decides the retry (RETRY_MAX_CASE_MS). These rows add no
|
||||
* case-count or model-work policy. The 10-second terms preserve the existing
|
||||
* Codex/recording finalization grace.
|
||||
/** Whole-file walls cover all existing cases, even if Bun runs them
|
||||
* sequentially. Mixed-tier files reserve their larger complete tier, never a
|
||||
* currently selected subset. caseMs is the longest single case budget, the
|
||||
* wall of one isolated case shard. These rows add no case-count or model-work
|
||||
* policy. The 10-second terms preserve the existing Codex/recording
|
||||
* finalization grace.
|
||||
*/
|
||||
export const FILE_RETRY_BUDGETS = [
|
||||
...STRICT_RETRY_CASE_BUDGETS,
|
||||
...[
|
||||
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-qa-callers.test.ts', attemptMs: 5 * (CAPTURE_MS + 15_000), caseMs: CAPTURE_MS + 15_000 },
|
||||
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
|
||||
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 4 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, caseMs: CAPTURE_LONG_MS },
|
||||
// Seventeen workflow judges include their 10s recording grace; the other
|
||||
// seven judges retain 120s. Supervise all 24 and the existing one retry.
|
||||
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 2 },
|
||||
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS, configuredRetries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000, configuredRetries: 1 },
|
||||
// seven judges retain 120s. Supervise all 24.
|
||||
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, caseMs: JUDGE_MS + 10_000 },
|
||||
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, caseMs: CAPTURE_MS },
|
||||
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 },
|
||||
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
|
||||
{ file: 'test/skill-e2e-plan-ceo-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
|
||||
{ file: 'test/skill-e2e-plan-eng-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
|
||||
{ file: 'test/skill-e2e-plan-design-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
|
||||
{ file: 'test/skill-e2e-plan-devex-finding-floor.test.ts', attemptMs: PTY_MS, caseMs: PTY_MS },
|
||||
{ file: 'test/skill-e2e-plan-mode-no-op.test.ts', attemptMs: 5 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
|
||||
{ file: 'test/skill-e2e-plan-ceo-mode-routing.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
|
||||
{ file: 'test/skill-e2e-plan-eng-plan-mode.test.ts', attemptMs: 2 * CAPTURE_LONG_MS, caseMs: CAPTURE_LONG_MS },
|
||||
{ file: 'test/skill-e2e-plan-prosons.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), caseMs: CAPTURE_MS + 10_000 },
|
||||
// Gate: six 300s cases + one 610s case; periodic: two 900s + three 600s.
|
||||
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS, configuredRetries: 1 },
|
||||
].map(({ file, attemptMs, caseMs, configuredRetries }) => {
|
||||
const retries = retriesWithinCaseCap(caseMs, configuredRetries);
|
||||
return {
|
||||
file, attemptMs, caseMs, retries,
|
||||
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
|
||||
};
|
||||
}),
|
||||
{ file: 'test/skill-e2e-plan.test.ts', attemptMs: Math.max(6 * CAPTURE_MS + CAPTURE_LONG_MS + 10_000, 2 * PTY_MS + 3 * CAPTURE_LONG_MS), caseMs: PTY_MS },
|
||||
].map(({ file, attemptMs, caseMs }) => ({
|
||||
file, attemptMs, caseMs,
|
||||
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
|
||||
shardReserveMs: SHARD_RESERVE_MS,
|
||||
shardMs: attemptMs + SHARD_RESERVE_MS,
|
||||
})),
|
||||
];
|
||||
|
||||
/** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */
|
||||
|
||||
@@ -14,8 +14,20 @@ import {
|
||||
formatComparison,
|
||||
generateCommentary,
|
||||
judgePassed,
|
||||
CONTRACT_VIOLATIONS_FILE,
|
||||
ContractViolation,
|
||||
TRIAL_ENV,
|
||||
TRIAL_OUTCOME_SCHEMA,
|
||||
expectContract,
|
||||
failureClassOf,
|
||||
formatTrialOutcomes,
|
||||
panelVerdict,
|
||||
parseTrialOutcomes,
|
||||
sanitizeTrialError,
|
||||
trialContextFromEnv,
|
||||
} from './eval-store';
|
||||
import type { EvalResult, EvalTestEntry, ComparisonResult } from './eval-store';
|
||||
import type { EvalResult, EvalTestEntry, ComparisonResult, PanelTrial, TrialOutcomeRecord } from './eval-store';
|
||||
import { EVAL_POLICY } from './periodic-exclude-data';
|
||||
import { manualReviewFixture } from './manual-judge-review-fixture';
|
||||
|
||||
let tmpDir: string;
|
||||
@@ -957,3 +969,191 @@ describe('generateCommentary', () => {
|
||||
expect(notes.some(n => n.includes('Stable run'))).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
// --- Trials, panel verdicts and contract vetoes (eval reliability policy) ---
|
||||
|
||||
const PANEL = EVAL_POLICY.panel;
|
||||
const pass = (trial: number, extra: Partial<PanelTrial> = {}): PanelTrial => ({ trial, outcome: 'passed', ...extra });
|
||||
const fail = (trial: number, extra: Partial<PanelTrial> = {}): PanelTrial => ({ trial, outcome: 'failed', ...extra });
|
||||
const behavior = (trials: PanelTrial[], quarantined = false) =>
|
||||
panelVerdict({ case: 'case-x', kind: 'behavior', panel: PANEL, trials, quarantined });
|
||||
|
||||
describe('panelVerdict', () => {
|
||||
test('policy constants are the approved pre-registration', () => {
|
||||
expect(EVAL_POLICY.panel).toEqual({ n: 3, k: 2 });
|
||||
expect(EVAL_POLICY.quarantine).toEqual({
|
||||
entry: { rate: 0.95, minTrials: 10 },
|
||||
exit: { rate: 0.97, minTrials: 10 },
|
||||
capFraction: 0.1,
|
||||
expiryWeeklyRuns: 8,
|
||||
});
|
||||
expect(EVAL_POLICY.infraRedispatch).toBe(1);
|
||||
});
|
||||
|
||||
test('rule: one trial, any failure fails the lane', () => {
|
||||
const ok = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 1, k: 1 }, trials: [pass(1)] });
|
||||
expect(ok).toMatchObject({ status: 'PASS', split: false, failsLane: false, coverage: true, marks: '✓' });
|
||||
const bad = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 1, k: 1 }, trials: [fail(1)] });
|
||||
expect(bad).toMatchObject({ status: 'FAIL', failsLane: true, coverage: false, redClass: 'VERDICT', marks: '✗' });
|
||||
});
|
||||
|
||||
test('behavior 3/3 is a clean PASS', () => {
|
||||
expect(behavior([pass(1), pass(2), pass(3)])).toMatchObject({ status: 'PASS', split: false, passed: 3, failsLane: false });
|
||||
});
|
||||
|
||||
test('behavior 2/3 is a split PASS that shows its failed trial', () => {
|
||||
const v = behavior([pass(1), fail(2, { exit_reason: 'timeout' }), pass(3)]);
|
||||
expect(v).toMatchObject({ status: 'PASS', split: true, passed: 2, failed: 1, failsLane: false, coverage: true, marks: '✓✗✓', reason: 'PASS 2/3' });
|
||||
expect(v.trials[1].exit_reason).toBe('timeout');
|
||||
});
|
||||
|
||||
test('behavior 1/3 and 0/3 fail the lane', () => {
|
||||
expect(behavior([pass(1), fail(2), fail(3)])).toMatchObject({ status: 'FAIL', failsLane: true, redClass: 'VERDICT' });
|
||||
expect(behavior([fail(1), fail(2), fail(3)])).toMatchObject({ status: 'FAIL', failsLane: true });
|
||||
});
|
||||
|
||||
test('a contract trial fails the panel even at 2/3', () => {
|
||||
const v = behavior([pass(1), pass(2), fail(3, { failure_class: 'contract' })]);
|
||||
expect(v).toMatchObject({ status: 'FAIL', contract: true, failsLane: true, redClass: 'VERDICT', reason: 'contract violation' });
|
||||
});
|
||||
|
||||
test('a missing trial is INCOMPLETE and fails the lane', () => {
|
||||
const v = behavior([pass(1), pass(3)]);
|
||||
expect(v).toMatchObject({ status: 'INCOMPLETE', failsLane: true, coverage: false, redClass: 'INCOMPLETE', marks: '✓·✓' });
|
||||
expect(v.reason).toContain('missing trial t2');
|
||||
});
|
||||
|
||||
test('duplicate or out-of-range trial records are INCOMPLETE, never deduplicated', () => {
|
||||
expect(behavior([pass(1), pass(2), pass(2), fail(3)]).status).toBe('INCOMPLETE');
|
||||
expect(behavior([pass(1), pass(2), pass(3), pass(4)]).reason).toContain('unexpected trial t4');
|
||||
const v = panelVerdict({ case: 'c', kind: 'behavior', panel: { n: 11, k: 6 }, trials: Array.from({ length: 10 }, (_, i) => pass(i + 2)) });
|
||||
expect(v.reason).toContain('missing trial t1');
|
||||
});
|
||||
|
||||
test('timeout and infra trials count as failed, never passing', () => {
|
||||
const v = behavior([pass(1), fail(2, { exit_reason: 'timeout' }), fail(3, { failure_class: 'infra' })]);
|
||||
expect(v).toMatchObject({ status: 'FAIL', passed: 1, failed: 2, failsLane: true, redClass: 'VERDICT' });
|
||||
const infra = behavior([pass(1), fail(2, { failure_class: 'infra' }), fail(3, { failure_class: 'infra' })]);
|
||||
expect(infra).toMatchObject({ status: 'FAIL', redClass: 'INFRA' });
|
||||
expect(failureClassOf({ exit_reason: 'timeout' })).toBe('timeout');
|
||||
expect(failureClassOf({})).toBe('assertion');
|
||||
});
|
||||
|
||||
test('quarantined: 1/3 does not fail the lane, 0/3 and contract do, no coverage credit', () => {
|
||||
expect(behavior([pass(1), fail(2), fail(3)], true)).toMatchObject({ status: 'FAIL', failsLane: false, coverage: false, redClass: null });
|
||||
expect(behavior([fail(1), fail(2), fail(3)], true)).toMatchObject({ status: 'FAIL', failsLane: true });
|
||||
expect(behavior([pass(1), pass(2), fail(3, { failure_class: 'contract' })], true)).toMatchObject({ status: 'FAIL', failsLane: true });
|
||||
expect(behavior([pass(1), pass(2), pass(3)], true)).toMatchObject({ status: 'PASS', coverage: false, failsLane: false });
|
||||
expect(behavior([pass(1), pass(2)], true)).toMatchObject({ status: 'INCOMPLETE', failsLane: true });
|
||||
});
|
||||
|
||||
test('quarantined rule keeps rule meaning (k = n)', () => {
|
||||
const v = panelVerdict({ case: 'r', kind: 'rule', panel: { n: 3, k: 3 }, trials: [pass(1), pass(2), fail(3)], quarantined: true });
|
||||
expect(v).toMatchObject({ status: 'FAIL', failsLane: false });
|
||||
});
|
||||
|
||||
test('all-skipped panel is SKIPPED with no credit; partly skipped is INCOMPLETE', () => {
|
||||
const skip = (trial: number): PanelTrial => ({ trial, outcome: 'skipped' });
|
||||
expect(behavior([skip(1), skip(2), skip(3)])).toMatchObject({ status: 'SKIPPED', coverage: false, failsLane: false });
|
||||
expect(behavior([pass(1), pass(2), skip(3)])).toMatchObject({ status: 'INCOMPLETE', failsLane: true });
|
||||
});
|
||||
|
||||
test('trials of different run attempts are never merged into one verdict', () => {
|
||||
expect(() => behavior([pass(1), pass(2), fail(3, { attempt: 2 })])).toThrow(/run attempts/);
|
||||
expect(behavior([pass(1, { attempt: 2 }), pass(2, { attempt: 2 }), pass(3, { attempt: 2 })]).attempt).toBe(2);
|
||||
});
|
||||
|
||||
test('invalid panels throw', () => {
|
||||
expect(() => panelVerdict({ case: 'c', kind: 'behavior', panel: { n: 3, k: 4 }, trials: [] })).toThrow(/invalid panel/);
|
||||
expect(() => panelVerdict({ case: 'c', kind: 'nope' as any, panel: { n: 1, k: 1 }, trials: [] })).toThrow(/unknown kind/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('trial context and expectContract', () => {
|
||||
let dir: string;
|
||||
const saved: Record<string, string | undefined> = {};
|
||||
const keys = [...Object.values(TRIAL_ENV), 'GSTACK_EVAL_DIR'];
|
||||
beforeEach(() => {
|
||||
dir = fs.mkdtempSync(path.join(os.tmpdir(), 'panel-verdict-'));
|
||||
for (const key of keys) saved[key] = process.env[key];
|
||||
});
|
||||
afterEach(() => {
|
||||
for (const key of keys) {
|
||||
if (saved[key] === undefined) delete process.env[key];
|
||||
else process.env[key] = saved[key];
|
||||
}
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
const setTrial = () => Object.assign(process.env, {
|
||||
[TRIAL_ENV.caseId]: 'case-x', [TRIAL_ENV.kind]: 'behavior', [TRIAL_ENV.trial]: '2',
|
||||
[TRIAL_ENV.panelN]: '3', [TRIAL_ENV.panelK]: '2', [TRIAL_ENV.policyVersion]: String(EVAL_POLICY.version),
|
||||
GSTACK_EVAL_DIR: dir,
|
||||
});
|
||||
|
||||
test('trialContextFromEnv: absent, complete, and malformed', () => {
|
||||
for (const key of Object.values(TRIAL_ENV)) delete process.env[key];
|
||||
expect(trialContextFromEnv()).toBeNull();
|
||||
setTrial();
|
||||
expect(trialContextFromEnv()).toEqual({ case_id: 'case-x', kind: 'behavior', trial: 2, panel: { n: 3, k: 2 }, policy_version: EVAL_POLICY.version });
|
||||
process.env[TRIAL_ENV.trial] = '4';
|
||||
expect(() => trialContextFromEnv()).toThrow(/Malformed trial context/);
|
||||
});
|
||||
|
||||
test('passing contract is a no-op', () => {
|
||||
setTrial();
|
||||
expect(() => expectContract(true, 'fine')).not.toThrow();
|
||||
expect(fs.existsSync(path.join(dir, CONTRACT_VIOLATIONS_FILE))).toBe(false);
|
||||
});
|
||||
|
||||
test('failed contract stamps the recorded entry and the sidecar before throwing', () => {
|
||||
setTrial();
|
||||
const collector = new EvalCollector('e2e', dir);
|
||||
collector.addTest({ name: 'case-x', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 });
|
||||
expect(() => expectContract(false, 'handoff missing', { collector, name: 'case-x' })).toThrow(ContractViolation);
|
||||
const partial = JSON.parse(fs.readFileSync(path.join(dir, '_partial-e2e.json'), 'utf-8'));
|
||||
expect(partial.tests[0]).toMatchObject({ passed: false, failure_class: 'contract', case_id: 'case-x', trial: 2, kind: 'behavior', panel: { n: 3, k: 2 } });
|
||||
const sidecar = fs.readFileSync(path.join(dir, CONTRACT_VIOLATIONS_FILE), 'utf-8').trim().split('\n').map((l) => JSON.parse(l));
|
||||
expect(sidecar).toEqual([expect.objectContaining({ case_id: 'case-x', trial: 2, message: 'handoff missing' })]);
|
||||
});
|
||||
|
||||
test('a contract marked before recording stamps the later record, or becomes its own at finalize', async () => {
|
||||
setTrial();
|
||||
const collector = new EvalCollector('e2e', dir);
|
||||
expect(() => expectContract(0, 'no question asked', { collector, name: 'later' })).toThrow('CONTRACT: no question asked');
|
||||
collector.addTest({ name: 'later', suite: 's', tier: 'e2e', passed: true, duration_ms: 1, cost_usd: 0 });
|
||||
expect(() => expectContract(null, 'never recorded', { collector, name: 'orphan' })).toThrow();
|
||||
const file = await collector.finalize();
|
||||
const tests = JSON.parse(fs.readFileSync(file, 'utf-8')).tests;
|
||||
expect(tests.find((t: any) => t.name === 'later')).toMatchObject({ passed: false, failure_class: 'contract' });
|
||||
expect(tests.find((t: any) => t.name === 'orphan')).toMatchObject({ passed: false, failure_class: 'contract', error: 'never recorded' });
|
||||
});
|
||||
});
|
||||
|
||||
describe('trial-outcomes JSONL', () => {
|
||||
const record = (extra: Partial<TrialOutcomeRecord> = {}): TrialOutcomeRecord => ({
|
||||
schema: TRIAL_OUTCOME_SCHEMA, case: 'case-x', file: 'test/x.test.ts', tier: 'gate', kind: 'behavior',
|
||||
trial: 1, panel: { n: 3, k: 2 }, attempt: 1, outcome: 'passed', duration_ms: 10, cost_usd: 0.1,
|
||||
policy_version: EVAL_POLICY.version, quarantined: false, execution: 'executed', source: 'shard', ...extra,
|
||||
});
|
||||
|
||||
test('round-trips valid records', () => {
|
||||
const records = [record(), record({ trial: 2, outcome: 'failed', failure_class: 'timeout', exit_reason: 'timeout' })];
|
||||
expect(parseTrialOutcomes(formatTrialOutcomes(records))).toEqual({ records, errors: [] });
|
||||
});
|
||||
|
||||
test('writer fails closed; reader reports bad lines as data errors', () => {
|
||||
expect(() => formatTrialOutcomes([record({ outcome: 'failed' })])).toThrow(/failed without failure_class/);
|
||||
expect(() => formatTrialOutcomes([record({ trial: 4 })])).toThrow(/trial invalid/);
|
||||
const text = `${JSON.stringify(record())}\nnot json\n${JSON.stringify({ ...record(), schema: 'other' })}\n`;
|
||||
const parsed = parseTrialOutcomes(text);
|
||||
expect(parsed.records).toHaveLength(1);
|
||||
expect(parsed.errors).toEqual(['line 2: not JSON', 'line 3: schema other']);
|
||||
expect(parseTrialOutcomes(text, { maxBytes: 10 }).errors[0]).toContain('exceed');
|
||||
});
|
||||
|
||||
test('sanitizeTrialError keeps one capped line without mentions', () => {
|
||||
expect(sanitizeTrialError('\n expected @garrytan to `see`\nsecond')).toBe("expected @\u200bgarrytan to 'see'");
|
||||
expect(sanitizeTrialError('x'.repeat(1000))!.length).toBe(300);
|
||||
expect(sanitizeTrialError('')).toBeUndefined();
|
||||
});
|
||||
});
|
||||
+367
-1
@@ -76,6 +76,18 @@ export interface EvalTestEntry {
|
||||
* its body again and re-records under the same name. Set by addTest. */
|
||||
attempt?: number;
|
||||
|
||||
// Trial identity (eval reliability policy). Stamped by addTest from the
|
||||
// TRIAL_ENV variables the paid runner sets on an isolated trial shard.
|
||||
/** Registry id (E2E_TIERS / LLM_JUDGE_TOUCHFILES key) this record belongs to. */
|
||||
case_id?: string;
|
||||
kind?: EvalCaseKind;
|
||||
/** 1-based trial index within the case's panel. */
|
||||
trial?: number;
|
||||
panel?: PanelShape;
|
||||
/** Why a failed record failed; 'contract' comes only from expectContract. */
|
||||
failure_class?: TrialFailureClass;
|
||||
policy_version?: number;
|
||||
|
||||
// E2E
|
||||
transcript?: any[];
|
||||
prompt?: string;
|
||||
@@ -131,6 +143,329 @@ export function evalEntryOutcome(entry: unknown): 'passed' | 'failed' | 'manual-
|
||||
return result.passed === true ? 'passed' : 'failed';
|
||||
}
|
||||
|
||||
// --- Trials and panel verdicts ---
|
||||
//
|
||||
// Paid evals never retry. Each case's kind (E2E_KINDS) fixes its trials before
|
||||
// the run; a panel verdict is computed once, by panelVerdict(), from exactly
|
||||
// panel.n trial records of one run attempt. The report, collector-outcomes,
|
||||
// the PR comment and pass-rates all read that one function.
|
||||
|
||||
export type EvalCaseKind = 'rule' | 'behavior' | 'judge';
|
||||
/** assertion: an ordinary failed expectation. contract: expectContract() fired
|
||||
* (fails the panel at any count). timeout: the case budget ran out.
|
||||
* infra: API/CLI/runner failure before the model could be graded. */
|
||||
export type TrialFailureClass = 'assertion' | 'contract' | 'timeout' | 'infra';
|
||||
export type TrialOutcome = 'passed' | 'failed' | 'skipped';
|
||||
export interface PanelShape { n: number; k: number }
|
||||
|
||||
/** Environment the paid runner sets on an isolated trial shard. */
|
||||
export const TRIAL_ENV = {
|
||||
caseId: 'GSTACK_EVAL_CASE_ID',
|
||||
kind: 'GSTACK_EVAL_KIND',
|
||||
trial: 'GSTACK_EVAL_TRIAL',
|
||||
panelN: 'GSTACK_EVAL_PANEL_N',
|
||||
panelK: 'GSTACK_EVAL_PANEL_K',
|
||||
policyVersion: 'GSTACK_EVAL_POLICY_VERSION',
|
||||
} as const;
|
||||
|
||||
/** Sidecar every expectContract() failure appends to (in GSTACK_EVAL_DIR), so a
|
||||
* contract veto survives a test that throws before recording its entry. */
|
||||
export const CONTRACT_VIOLATIONS_FILE = 'contract-violations.jsonl';
|
||||
|
||||
export interface TrialContext {
|
||||
case_id: string;
|
||||
kind: EvalCaseKind;
|
||||
trial: number;
|
||||
panel: PanelShape;
|
||||
policy_version: number;
|
||||
}
|
||||
|
||||
const EVAL_KINDS: readonly EvalCaseKind[] = ['rule', 'behavior', 'judge'];
|
||||
const FAILURE_CLASSES: readonly TrialFailureClass[] = ['assertion', 'contract', 'timeout', 'infra'];
|
||||
|
||||
function positiveInt(raw: string | undefined): number | null {
|
||||
if (raw === undefined || !/^[1-9][0-9]*$/.test(raw)) return null;
|
||||
return Number(raw);
|
||||
}
|
||||
|
||||
/** Trial context of this process, or null outside an isolated trial shard.
|
||||
* A partial or malformed context throws: a mislabeled trial is fail-open. */
|
||||
export function trialContextFromEnv(env: NodeJS.ProcessEnv = process.env): TrialContext | null {
|
||||
const caseId = env[TRIAL_ENV.caseId];
|
||||
if (!caseId) return null;
|
||||
const kind = env[TRIAL_ENV.kind] as EvalCaseKind | undefined;
|
||||
const trial = positiveInt(env[TRIAL_ENV.trial]);
|
||||
const n = positiveInt(env[TRIAL_ENV.panelN]);
|
||||
const k = positiveInt(env[TRIAL_ENV.panelK]);
|
||||
const policy = positiveInt(env[TRIAL_ENV.policyVersion]);
|
||||
if (!kind || !EVAL_KINDS.includes(kind) || trial === null || n === null || k === null || policy === null || k > n || trial > n) {
|
||||
throw new Error(`Malformed trial context for ${caseId}: ${Object.values(TRIAL_ENV).map((name) => `${name}=${env[name] ?? ''}`).join(' ')}`);
|
||||
}
|
||||
return { case_id: caseId, kind, trial, panel: { n, k }, policy_version: policy };
|
||||
}
|
||||
|
||||
/** Failure class of a failed record: an explicit class wins, then the exit reason. */
|
||||
export function failureClassOf(entry: Pick<EvalTestEntry, 'failure_class' | 'exit_reason'>): TrialFailureClass {
|
||||
if (entry.failure_class && FAILURE_CLASSES.includes(entry.failure_class)) return entry.failure_class;
|
||||
return entry.exit_reason === 'timeout' ? 'timeout' : 'assertion';
|
||||
}
|
||||
|
||||
export class ContractViolation extends Error {
|
||||
constructor(message: string) {
|
||||
super(`CONTRACT: ${message}`);
|
||||
this.name = 'ContractViolation';
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Assert a contract: an outcome the product must meet on every run. On failure
|
||||
* it records failure_class 'contract' before throwing, both on the collector
|
||||
* entry named `record.name` (now or when the test records it) and in the
|
||||
* GSTACK_EVAL_DIR sidecar, so panelVerdict() fails the panel even at 2 of 3.
|
||||
*/
|
||||
export function expectContract(
|
||||
condition: unknown,
|
||||
message: string,
|
||||
record?: { collector: EvalCollector | null; name: string },
|
||||
): asserts condition {
|
||||
if (condition) return;
|
||||
record?.collector?.markContractViolation(record.name, message);
|
||||
const evalDir = process.env.GSTACK_EVAL_DIR;
|
||||
if (evalDir) {
|
||||
const context = trialContextFromEnv();
|
||||
fs.mkdirSync(evalDir, { recursive: true });
|
||||
fs.appendFileSync(path.join(evalDir, CONTRACT_VIOLATIONS_FILE), JSON.stringify({
|
||||
case_id: context?.case_id ?? record?.name ?? null,
|
||||
name: record?.name ?? null,
|
||||
trial: context?.trial ?? null,
|
||||
message,
|
||||
at: new Date().toISOString(),
|
||||
}) + '\n');
|
||||
}
|
||||
throw new ContractViolation(message);
|
||||
}
|
||||
|
||||
export interface PanelTrial {
|
||||
trial: number;
|
||||
outcome: TrialOutcome;
|
||||
/** Required meaning for a failed trial; absent reads as 'assertion'. */
|
||||
failure_class?: TrialFailureClass;
|
||||
/** CI run attempt (github.run_attempt); absent means 1. */
|
||||
attempt?: number;
|
||||
exit_reason?: string;
|
||||
error?: string;
|
||||
execution?: 'executed' | 'reused';
|
||||
}
|
||||
|
||||
export interface PanelVerdictInput {
|
||||
case: string;
|
||||
kind: EvalCaseKind;
|
||||
panel: PanelShape;
|
||||
trials: readonly PanelTrial[];
|
||||
quarantined?: boolean;
|
||||
}
|
||||
|
||||
export type PanelStatus = 'PASS' | 'FAIL' | 'INCOMPLETE' | 'SKIPPED';
|
||||
|
||||
export interface PanelVerdict {
|
||||
case: string;
|
||||
kind: EvalCaseKind;
|
||||
panel: PanelShape;
|
||||
attempt: number;
|
||||
quarantined: boolean;
|
||||
status: PanelStatus;
|
||||
passed: number;
|
||||
failed: number;
|
||||
/** A failed trial carried failure_class 'contract'. */
|
||||
contract: boolean;
|
||||
/** PASS with at least one failed trial: shown as `PASS k/n`, never clean. */
|
||||
split: boolean;
|
||||
/** Whether this verdict makes the lane red. */
|
||||
failsLane: boolean;
|
||||
/** Whether it counts as passing coverage (never for quarantined or skipped). */
|
||||
coverage: boolean;
|
||||
/** Machine classification of a lane-failing verdict: INCOMPLETE (missing or
|
||||
* malformed trial records), INFRA (every failed trial is infra-class), or
|
||||
* VERDICT (a real red). Null when the verdict does not fail the lane. */
|
||||
redClass: 'INCOMPLETE' | 'INFRA' | 'VERDICT' | null;
|
||||
/** One glyph per trial index: ✓ pass, ✗ fail, – skipped, · missing. */
|
||||
marks: string;
|
||||
reason: string;
|
||||
trials: PanelTrial[];
|
||||
}
|
||||
|
||||
/**
|
||||
* The single verdict function. `rule`/`judge` cases run panel {1,1}; `behavior`
|
||||
* cases run EVAL_POLICY.panel; a quarantined case runs a full panel whose k
|
||||
* keeps its kind's meaning (k = n for rule). Verdict: INCOMPLETE unless
|
||||
* exactly one record per trial index 1..n; SKIPPED when every trial skipped;
|
||||
* FAIL on any contract trial; otherwise PASS iff passed >= k. A quarantined
|
||||
* FAIL fails the lane only on a hard break (0 of n) or a contract violation.
|
||||
*/
|
||||
export function panelVerdict(input: PanelVerdictInput): PanelVerdict {
|
||||
const { n, k } = input.panel;
|
||||
if (!Number.isInteger(n) || !Number.isInteger(k) || n < 1 || k < 1 || k > n) {
|
||||
throw new Error(`${input.case}: invalid panel {n:${n}, k:${k}}`);
|
||||
}
|
||||
if (!EVAL_KINDS.includes(input.kind)) throw new Error(`${input.case}: unknown kind ${String(input.kind)}`);
|
||||
const attempts = new Set(input.trials.map((t) => t.attempt ?? 1));
|
||||
if (attempts.size > 1) {
|
||||
throw new Error(`${input.case}: trials from run attempts ${[...attempts].join(', ')}; compute one verdict per attempt`);
|
||||
}
|
||||
const attempt = [...attempts][0] ?? 1;
|
||||
const quarantined = input.quarantined === true;
|
||||
const trials = [...input.trials].sort((a, b) => a.trial - b.trial);
|
||||
const byIndex = new Map<number, PanelTrial>();
|
||||
const problems: string[] = [];
|
||||
for (const t of trials) {
|
||||
if (!Number.isInteger(t.trial) || t.trial < 1 || t.trial > n) problems.push(`unexpected trial t${t.trial}`);
|
||||
else if (byIndex.has(t.trial)) problems.push(`duplicate trial t${t.trial}`);
|
||||
else if (t.outcome !== 'passed' && t.outcome !== 'failed' && t.outcome !== 'skipped') problems.push(`t${t.trial} has outcome ${String(t.outcome)}`);
|
||||
else byIndex.set(t.trial, t);
|
||||
}
|
||||
for (let i = 1; i <= n; i++) if (!trials.some((t) => t.trial === i)) problems.push(`missing trial t${i}`);
|
||||
const marks = Array.from({ length: n }, (_, i) => {
|
||||
const t = byIndex.get(i + 1);
|
||||
return !t ? '·' : t.outcome === 'passed' ? '✓' : t.outcome === 'failed' ? '✗' : '–';
|
||||
}).join('');
|
||||
const passed = [...byIndex.values()].filter((t) => t.outcome === 'passed').length;
|
||||
const failedTrials = [...byIndex.values()].filter((t) => t.outcome === 'failed');
|
||||
const skipped = [...byIndex.values()].filter((t) => t.outcome === 'skipped').length;
|
||||
const contract = failedTrials.some((t) => failureClassOf(t) === 'contract');
|
||||
const base = { case: input.case, kind: input.kind, panel: { n, k }, attempt, quarantined, passed, failed: failedTrials.length, contract, marks, trials };
|
||||
|
||||
if (problems.length === 0 && skipped === n) {
|
||||
return { ...base, status: 'SKIPPED', split: false, failsLane: false, coverage: false, redClass: null, reason: 'every trial skipped (no verdict credit)' };
|
||||
}
|
||||
if (problems.length === 0 && skipped > 0) problems.push(`${skipped} of ${n} trials skipped`);
|
||||
if (problems.length > 0) {
|
||||
return { ...base, status: 'INCOMPLETE', split: false, failsLane: true, coverage: false, redClass: 'INCOMPLETE', reason: problems.join('; ') };
|
||||
}
|
||||
if (!contract && passed >= k) {
|
||||
const split = failedTrials.length > 0;
|
||||
return {
|
||||
...base, status: 'PASS', split, failsLane: false, coverage: !quarantined, redClass: null,
|
||||
reason: split ? `PASS ${passed}/${n}` : `${passed}/${n} passed`,
|
||||
};
|
||||
}
|
||||
const hardBreak = passed === 0;
|
||||
const failsLane = !quarantined || contract || hardBreak;
|
||||
const allInfra = !contract && failedTrials.length > 0 && failedTrials.every((t) => failureClassOf(t) === 'infra');
|
||||
const why = contract ? 'contract violation' : `${passed}/${n} passed, needs ${k}`;
|
||||
return {
|
||||
...base, status: 'FAIL', split: false, failsLane, coverage: false,
|
||||
redClass: failsLane ? (allInfra ? 'INFRA' : 'VERDICT') : null,
|
||||
reason: !quarantined ? why
|
||||
: contract ? `${why}; quarantine never excuses a contract`
|
||||
: hardBreak ? `${why}; quarantined hard break`
|
||||
: `${why}; quarantined, does not fail the lane`,
|
||||
};
|
||||
}
|
||||
|
||||
// --- trial-outcomes JSONL (one line per trial; pass-rate history input) ---
|
||||
|
||||
export const TRIAL_OUTCOME_SCHEMA = 'gstack-trial-outcome/v1';
|
||||
export const TRIAL_OUTCOMES_FILE = 'trial-outcomes.jsonl';
|
||||
/** Cap on a stored `error` line (sanitized first line of the failure). */
|
||||
export const TRIAL_ERROR_MAX = 300;
|
||||
|
||||
export interface TrialOutcomeRecord {
|
||||
schema: typeof TRIAL_OUTCOME_SCHEMA;
|
||||
/** Registry id. */
|
||||
case: string;
|
||||
file: string;
|
||||
tier: string;
|
||||
kind: EvalCaseKind;
|
||||
trial: number;
|
||||
panel: PanelShape;
|
||||
/** CI run attempt (github.run_attempt); 1 locally and for pre-policy backfill. */
|
||||
attempt: number;
|
||||
outcome: TrialOutcome;
|
||||
/** Present exactly when outcome is 'failed'. */
|
||||
failure_class?: TrialFailureClass;
|
||||
exit_reason?: string;
|
||||
error?: string;
|
||||
duration_ms: number;
|
||||
cost_usd: number;
|
||||
model?: string;
|
||||
cli_version?: string;
|
||||
/** Reuse input key of the trial's shard, when known. */
|
||||
input_identity?: string;
|
||||
/** EVAL_POLICY.version; 0 marks pre-policy backfill. */
|
||||
policy_version: number;
|
||||
quarantined: boolean;
|
||||
execution: 'executed' | 'reused';
|
||||
/** shard: isolated trial shard status. junit: a rule file shard's per-test
|
||||
* JUnit outcome. backfill: imported pre-policy artifact record. */
|
||||
source: 'shard' | 'junit' | 'backfill';
|
||||
run_id?: string;
|
||||
sha?: string;
|
||||
lane?: string;
|
||||
recorded_at?: string;
|
||||
/** History series key: a hash of the case's own touchfiles (GLOBAL_TOUCHFILES excluded), stamped by the report job. */
|
||||
series_identity?: string;
|
||||
}
|
||||
|
||||
/** First line of free text, stripped of @-mentions and control characters, capped. */
|
||||
export function sanitizeTrialError(text: string | undefined): string | undefined {
|
||||
if (!text) return undefined;
|
||||
const first = text.split('\n').map((l) => l.trim()).find((l) => l.length > 0);
|
||||
if (!first) return undefined;
|
||||
// eslint-disable-next-line no-control-regex
|
||||
const clean = first.replace(/[\u0000-\u001f\u007f]/g, ' ').replace(/`/g, "'").replace(/@(?=[A-Za-z0-9_-])/g, '@\u200b');
|
||||
return clean.length > TRIAL_ERROR_MAX ? `${clean.slice(0, TRIAL_ERROR_MAX - 1)}…` : clean;
|
||||
}
|
||||
|
||||
function trialRecordProblems(r: any): string[] {
|
||||
const problems: string[] = [];
|
||||
if (!r || typeof r !== 'object' || Array.isArray(r)) return ['not an object'];
|
||||
if (r.schema !== TRIAL_OUTCOME_SCHEMA) problems.push(`schema ${String(r.schema)}`);
|
||||
for (const key of ['case', 'file', 'tier'] as const) if (typeof r[key] !== 'string' || r[key].length === 0) problems.push(`${key} missing`);
|
||||
if (!EVAL_KINDS.includes(r.kind)) problems.push(`kind ${String(r.kind)}`);
|
||||
const n = r.panel?.n, k = r.panel?.k;
|
||||
if (!Number.isInteger(n) || !Number.isInteger(k) || n < 1 || k < 1 || k > n) problems.push('panel invalid');
|
||||
if (!Number.isInteger(r.trial) || r.trial < 1 || (Number.isInteger(n) && r.trial > n)) problems.push('trial invalid');
|
||||
if (!Number.isInteger(r.attempt) || r.attempt < 1) problems.push('attempt invalid');
|
||||
if (!['passed', 'failed', 'skipped'].includes(r.outcome)) problems.push(`outcome ${String(r.outcome)}`);
|
||||
if (r.outcome === 'failed' && !FAILURE_CLASSES.includes(r.failure_class)) problems.push('failed without failure_class');
|
||||
if (r.outcome !== 'failed' && r.failure_class !== undefined) problems.push('failure_class on a non-failed trial');
|
||||
if (typeof r.duration_ms !== 'number' || !Number.isFinite(r.duration_ms) || r.duration_ms < 0) problems.push('duration_ms invalid');
|
||||
if (typeof r.cost_usd !== 'number' || !Number.isFinite(r.cost_usd) || r.cost_usd < 0) problems.push('cost_usd invalid');
|
||||
if (!Number.isInteger(r.policy_version) || r.policy_version < 0) problems.push('policy_version invalid');
|
||||
if (typeof r.quarantined !== 'boolean') problems.push('quarantined invalid');
|
||||
if (r.execution !== 'executed' && r.execution !== 'reused') problems.push('execution invalid');
|
||||
if (!['shard', 'junit', 'backfill'].includes(r.source)) problems.push('source invalid');
|
||||
if (r.error !== undefined && (typeof r.error !== 'string' || r.error.length > TRIAL_ERROR_MAX)) problems.push('error invalid');
|
||||
if (r.series_identity !== undefined && (typeof r.series_identity !== 'string' || !/^[\w.-]{1,64}$/.test(r.series_identity))) problems.push('series_identity invalid');
|
||||
return problems;
|
||||
}
|
||||
|
||||
/** Serialize records as JSONL; throws on any invalid record (writers fail closed). */
|
||||
export function formatTrialOutcomes(records: readonly TrialOutcomeRecord[]): string {
|
||||
return records.map((r) => {
|
||||
const problems = trialRecordProblems(r);
|
||||
if (problems.length > 0) throw new Error(`invalid trial record ${r?.case}~t${r?.trial}: ${problems.join(', ')}`);
|
||||
return JSON.stringify(r);
|
||||
}).join('\n') + (records.length > 0 ? '\n' : '');
|
||||
}
|
||||
|
||||
/** Parse downloaded JSONL as data only: invalid lines are reported, never guessed. */
|
||||
export function parseTrialOutcomes(text: string, opts: { maxBytes?: number } = {}): { records: TrialOutcomeRecord[]; errors: string[] } {
|
||||
const maxBytes = opts.maxBytes ?? 16 * 1024 * 1024;
|
||||
if (Buffer.byteLength(text) > maxBytes) return { records: [], errors: [`trial outcomes exceed ${maxBytes} bytes`] };
|
||||
const records: TrialOutcomeRecord[] = [];
|
||||
const errors: string[] = [];
|
||||
text.split('\n').forEach((line, i) => {
|
||||
if (line.trim() === '') return;
|
||||
let parsed: unknown;
|
||||
try { parsed = JSON.parse(line); } catch { errors.push(`line ${i + 1}: not JSON`); return; }
|
||||
const problems = trialRecordProblems(parsed);
|
||||
if (problems.length > 0) errors.push(`line ${i + 1}: ${problems.join(', ')}`);
|
||||
else records.push(parsed as TrialOutcomeRecord);
|
||||
});
|
||||
return { records, errors };
|
||||
}
|
||||
|
||||
export interface EvalResult {
|
||||
schema_version: number;
|
||||
version: string;
|
||||
@@ -887,6 +1222,7 @@ export class EvalCollector {
|
||||
private shard: string | null;
|
||||
private fileNamespace?: string;
|
||||
private createdAt = Date.now();
|
||||
private pendingContract = new Map<string, string>();
|
||||
|
||||
constructor(tier: 'e2e' | 'llm-judge', evalDir?: string, fileNamespace?: string) {
|
||||
if (fileNamespace !== undefined && !/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(fileNamespace)) {
|
||||
@@ -903,7 +1239,29 @@ export class EvalCollector {
|
||||
// names are unique by convention). Stamp the 1-based attempt so a
|
||||
// pass-on-attempt-2 stays visible forever — the stream hides it.
|
||||
const prior = this.tests.filter((t) => t.name === entry.name).length;
|
||||
this.tests.push({ ...entry, attempt: prior + 1 });
|
||||
const context = trialContextFromEnv();
|
||||
const record: EvalTestEntry = { ...(context ?? {}), ...entry, attempt: prior + 1 };
|
||||
const contract = this.pendingContract.get(entry.name);
|
||||
if (contract !== undefined) {
|
||||
this.pendingContract.delete(entry.name);
|
||||
Object.assign(record, { passed: false, failure_class: 'contract', error: record.error ?? contract });
|
||||
}
|
||||
this.tests.push(record);
|
||||
this.savePartial();
|
||||
}
|
||||
|
||||
/** expectContract() hook: mark `name`'s latest record (or its next one) as a
|
||||
* contract failure. An unmatched mark becomes its own failed record at
|
||||
* finalize, so the veto is never lost. */
|
||||
markContractViolation(name: string, message: string): void {
|
||||
const existing = this.tests.filter((t) => t.name === name).at(-1);
|
||||
if (!existing) {
|
||||
this.pendingContract.set(name, message);
|
||||
return;
|
||||
}
|
||||
existing.passed = false;
|
||||
existing.failure_class = 'contract';
|
||||
existing.error = existing.error ?? message;
|
||||
this.savePartial();
|
||||
}
|
||||
|
||||
@@ -959,6 +1317,14 @@ export class EvalCollector {
|
||||
async finalize(): Promise<string> {
|
||||
if (this.finalized) return '';
|
||||
this.finalized = true;
|
||||
for (const [name, message] of this.pendingContract) {
|
||||
this.tests.push({
|
||||
...(trialContextFromEnv() ?? {}),
|
||||
name, suite: 'contract', tier: this.tier, passed: false, duration_ms: 0, cost_usd: 0,
|
||||
failure_class: 'contract', error: message, attempt: 1,
|
||||
});
|
||||
}
|
||||
this.pendingContract.clear();
|
||||
|
||||
const git = getGitInfo();
|
||||
const version = getVersion();
|
||||
|
||||
@@ -23,6 +23,8 @@ export interface JudgeScore {
|
||||
reasoning: string;
|
||||
}
|
||||
|
||||
export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const;
|
||||
|
||||
export interface JudgeRefusalEvidence {
|
||||
stop_reason: 'refusal';
|
||||
response_id: string | null;
|
||||
@@ -196,6 +198,63 @@ export async function callJudge<T>(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this
|
||||
* helper (imported by many paid tests) does not pull the quarantine registry
|
||||
* into their touchfile closure. test/judge-panel.test.ts pins the two equal.
|
||||
*/
|
||||
export const JUDGE_PANEL_SAMPLES = 3;
|
||||
|
||||
/**
|
||||
* Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of
|
||||
* independent samples of the SAME prompt concurrently, inside its unchanged
|
||||
* JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
|
||||
* against the unchanged minimum; boolean fields gate on a strict majority.
|
||||
* A sample that errors (refusal, truncation, non-JSON, malformed field) fails
|
||||
* the whole panel and is never resampled. callJudge's 429 backoff happens
|
||||
* before any model output exists, so it is transport, not a verdict retry.
|
||||
*/
|
||||
export async function judgePanel<T>(sample: () => Promise<T>): Promise<T[]> {
|
||||
const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample()));
|
||||
const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []);
|
||||
if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult<T>).value);
|
||||
const first = failures[0]!;
|
||||
// A refusal is an unscored panel only when EVERY sample refused; a partial
|
||||
// refusal beside scored samples is an ordinary failed panel.
|
||||
if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) {
|
||||
throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`);
|
||||
}
|
||||
throw first.reason;
|
||||
}
|
||||
|
||||
/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */
|
||||
export function judgePanelMean<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, keys: readonly K[]): Record<K, number> {
|
||||
if (samples.length === 0) throw new Error('Judge panel has no samples');
|
||||
return Object.fromEntries(keys.map(key => {
|
||||
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
|
||||
const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value));
|
||||
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`);
|
||||
return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length];
|
||||
})) as Record<K, number>;
|
||||
}
|
||||
|
||||
/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */
|
||||
export function judgePanelMajority<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, key: K): boolean {
|
||||
if (samples.length === 0) throw new Error('Judge panel has no samples');
|
||||
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
|
||||
const bad = values.findIndex(value => typeof value !== 'boolean');
|
||||
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`);
|
||||
return values.filter(value => value === true).length * 2 > values.length;
|
||||
}
|
||||
|
||||
/** Sample reasoning lines, numbered, for the collector record. */
|
||||
export function judgePanelReasoning(samples: ReadonlyArray<unknown>): string {
|
||||
return samples.map((sample, index) => {
|
||||
const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined;
|
||||
return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`;
|
||||
}).join('\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Score documentation quality on clarity/completeness/actionability (1-5).
|
||||
*/
|
||||
|
||||
@@ -59,3 +59,67 @@ export const CASE_CI_EXCLUDE: Record<string, { reason: string; tracking: string
|
||||
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
* Paid-eval verdict policy, pre-registered (approved 2026-09-29). Frozen before
|
||||
* the census: any change after seeing census results needs Garry's
|
||||
* re-approval and a fresh census, and bumps `version` (every trial record
|
||||
* carries it as policy_version, so pass-rate history segments at the change).
|
||||
* panel - behavior cases and quarantined cases run n independent
|
||||
* trials; a behavior panel PASSES at >= k passing trials with
|
||||
* no contract violation. Rule and judge cases run one trial.
|
||||
* quarantine - entry below `entry.rate` per trial over >= `entry.minTrials`
|
||||
* new-policy trials; exit at >= `exit.rate` over >=
|
||||
* `exit.minTrials`; at most `capFraction` of each tier's
|
||||
* blocking cases; an entry expires after `expiryWeeklyRuns`.
|
||||
* judge - a judge case draws `samples` independent samples of one
|
||||
* prompt concurrently; numeric dimensions gate on the panel
|
||||
* mean against the unchanged threshold, booleans on a strict
|
||||
* majority; an erroring sample fails the panel, never resampled.
|
||||
* drift - one-sided Fisher exact alarm between input-identity series
|
||||
* (Holm-controlled across the cases tested in one report).
|
||||
* infraRedispatch - a census whose every red verdict is machine-classified
|
||||
* INFRA or INCOMPLETE may be re-dispatched this many times as
|
||||
* a new run; both runs are reported.
|
||||
*/
|
||||
export const EVAL_POLICY = {
|
||||
version: 1,
|
||||
panel: { n: 3, k: 2 },
|
||||
quarantine: {
|
||||
entry: { rate: 0.95, minTrials: 10 },
|
||||
exit: { rate: 0.97, minTrials: 10 },
|
||||
capFraction: 0.10,
|
||||
expiryWeeklyRuns: 8,
|
||||
},
|
||||
judge: { samples: 3 },
|
||||
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
|
||||
infraRedispatch: 1,
|
||||
} as const;
|
||||
|
||||
/**
|
||||
* Quarantined paid cases, keyed by registry id (an E2E_TIERS key). A
|
||||
* quarantined case still runs its full panel and reports in every lane, but
|
||||
* its failed verdict cannot fail the lane unless the panel is a hard break
|
||||
* (0 of n) or a trial violated a contract; it never counts as passing
|
||||
* coverage. An entry needs the entry rule met on the current input identity,
|
||||
* a written diagnosis that the failures are detector, harness or model-latency
|
||||
* failures (a product defect is never quarantined), and unchanged case
|
||||
* touchfiles in the change that adds it. Pinned by
|
||||
* test/periodic-exclude-policy.test.ts.
|
||||
* reason - the written diagnosis, with the pass-rate evidence
|
||||
* failureClass - what the diagnosis found; a product defect has no class here
|
||||
* tracking - issue or TODOS pointer
|
||||
* owner - who removes it
|
||||
* enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here)
|
||||
* exit - the measurable exit condition
|
||||
* At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and
|
||||
* periodic are the blocking tiers) may be quarantined at once.
|
||||
*/
|
||||
export const CASE_QUARANTINE: Record<string, {
|
||||
reason: string;
|
||||
failureClass: 'detector' | 'harness' | 'model-latency';
|
||||
tracking: string;
|
||||
owner: string;
|
||||
enteredAt: string;
|
||||
exit: string;
|
||||
}> = {};
|
||||
@@ -34,6 +34,8 @@ import {
|
||||
E2E_TIERS,
|
||||
LLM_JUDGE_TOUCHFILES,
|
||||
GLOBAL_TOUCHFILES,
|
||||
E2E_KINDS,
|
||||
BEHAVIOR_WHY,
|
||||
} from './touchfiles-data';
|
||||
|
||||
/** Repo-relative path of the pure-data file (the map-diff subject). */
|
||||
@@ -145,6 +147,9 @@ export interface TouchfileMaps {
|
||||
E2E_TIERS: Record<string, string>;
|
||||
LLM_JUDGE_TOUCHFILES: Record<string, string[]>;
|
||||
GLOBAL_TOUCHFILES: string[];
|
||||
/** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */
|
||||
E2E_KINDS?: Record<string, string>;
|
||||
BEHAVIOR_WHY?: Record<string, string>;
|
||||
}
|
||||
|
||||
export type MapDiffCause =
|
||||
@@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = {
|
||||
E2E_TIERS,
|
||||
LLM_JUDGE_TOUCHFILES,
|
||||
GLOBAL_TOUCHFILES,
|
||||
E2E_KINDS,
|
||||
BEHAVIOR_WHY,
|
||||
};
|
||||
|
||||
function isStringArray(v: unknown): v is string[] {
|
||||
@@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps {
|
||||
return isRecordOfStringArrays(o.E2E_TOUCHFILES)
|
||||
&& isRecordOfStrings(o.E2E_TIERS)
|
||||
&& isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES)
|
||||
&& isStringArray(o.GLOBAL_TOUCHFILES);
|
||||
&& isStringArray(o.GLOBAL_TOUCHFILES)
|
||||
&& (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS))
|
||||
&& (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY));
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure map-diff core (injectable for tests — no git, no filesystem).
|
||||
*
|
||||
* A key counts as CHANGED when it was added to any per-key map, its dep-list
|
||||
* array differs, or its tier value flipped. A key counts as REMOVED only when
|
||||
* array differs, or its tier, kind or behavior tolerance changed. A per-key
|
||||
* map missing on the old side (a base revision older than E2E_KINDS /
|
||||
* BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when
|
||||
* it is gone from every new per-key map; a key dropped from one map but still
|
||||
* present in another (e.g. tier entry deleted, touchfile entry kept) counts
|
||||
* as changed — conservative, because the test still exists with a different
|
||||
@@ -212,7 +223,7 @@ export function diffTouchfileMapsCore(
|
||||
oldMaps: TouchfileMaps,
|
||||
newMaps: TouchfileMaps,
|
||||
): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } {
|
||||
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const;
|
||||
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const;
|
||||
const changed = new Set<string>();
|
||||
const rawRemoved = new Set<string>();
|
||||
|
||||
@@ -290,6 +301,8 @@ export function diffTouchfileMaps(
|
||||
' E2E_TIERS: m.E2E_TIERS,',
|
||||
' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,',
|
||||
' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,',
|
||||
' E2E_KINDS: m.E2E_KINDS,',
|
||||
' BEHAVIOR_WHY: m.BEHAVIOR_WHY,',
|
||||
'}));',
|
||||
'',
|
||||
].join('\n'));
|
||||
|
||||
@@ -1573,3 +1573,306 @@ export const GLOBAL_TOUCHFILES = [
|
||||
// diffed per key, so a data-only edit runs just the affected tests.
|
||||
// Map-diff fails CLOSED — any error on that path still runs everything.
|
||||
];
|
||||
|
||||
/**
|
||||
* Eval kind per live case (every E2E_TIERS and LLM_JUDGE_TOUCHFILES key).
|
||||
* The kind fixes the trial policy before the run (EVAL_POLICY in
|
||||
* periodic-exclude-data.ts):
|
||||
* rule - one trial; any failed assertion fails the verdict. The default.
|
||||
* behavior - a panel of independent trials, PASS at the policy majority;
|
||||
* needs a BEHAVIOR_WHY entry naming the tolerated deviation.
|
||||
* judge - an LLM-judge score of a static input, sampled as a panel.
|
||||
* Reclassification is a reviewed diff, never a runtime switch.
|
||||
*/
|
||||
export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'ship-skipped-queued-finding': 'rule',
|
||||
'investigate-owned-completion': 'rule',
|
||||
'investigate-owned-abort': 'rule',
|
||||
'investigate-owned-ending-error': 'rule',
|
||||
'shared-libs-review-path-eligibility': 'rule',
|
||||
'shared-libs-review-index-flags': 'rule',
|
||||
'shared-libs-review-prior-coverage': 'rule',
|
||||
'shared-libs-codex-read-only': 'rule',
|
||||
'shared-libs-read-only': 'rule',
|
||||
'shared-libs-unsupported-git': 'rule',
|
||||
'shared-libs-review-lifecycle': 'rule',
|
||||
'shared-libs-review-revalidation': 'rule',
|
||||
'shared-libs-opportunity-judgment': 'behavior',
|
||||
'shared-libs-pr-coverage': 'rule',
|
||||
'shared-libs-plan-callers': 'rule',
|
||||
'browse-basic': 'rule',
|
||||
'browse-snapshot': 'rule',
|
||||
'aside-browse-basic': 'rule',
|
||||
'aside-browse-flow': 'rule',
|
||||
'aside-qa-quick': 'rule',
|
||||
'aside-scrape-json': 'rule',
|
||||
'aside-canary-quick': 'rule',
|
||||
'hermetic-canary': 'rule',
|
||||
'hermetic-sentinel': 'rule',
|
||||
'skillmd-setup-discovery': 'rule',
|
||||
'skillmd-no-local-binary': 'rule',
|
||||
'skillmd-outside-git': 'rule',
|
||||
'session-awareness': 'rule',
|
||||
'operational-learning': 'rule',
|
||||
'first-task-scaffold': 'rule',
|
||||
'qa-quick': 'rule',
|
||||
'qa-b6-static': 'rule',
|
||||
'qa-b7-spa': 'rule',
|
||||
'qa-b8-checkout': 'rule',
|
||||
'qa-only-no-fix': 'rule',
|
||||
'qa-fix-loop': 'rule',
|
||||
'qa-bootstrap': 'rule',
|
||||
'review-exploratory-small-cli': 'rule',
|
||||
'ship-exploratory-small-cli': 'rule',
|
||||
'ship-exploratory-unavailable': 'rule',
|
||||
'ship-exploratory-plan-checks': 'rule',
|
||||
'ship-exploratory-late-input': 'rule',
|
||||
'qa-functional-cli-report': 'rule',
|
||||
'qa-functional-webhook-report': 'rule',
|
||||
'qa-functional-cli-fix': 'rule',
|
||||
'qa-functional-webhook-fix': 'rule',
|
||||
'review-sql-injection': 'rule',
|
||||
'review-enum-completeness': 'rule',
|
||||
'review-base-branch': 'rule',
|
||||
'review-design-lite': 'behavior',
|
||||
'review-coverage-audit': 'rule',
|
||||
'review-dashboard-via': 'rule',
|
||||
'review-army-migration-safety': 'rule',
|
||||
'review-army-perf-n-plus-one': 'rule',
|
||||
'review-army-delivery-audit': 'rule',
|
||||
'review-army-quality-score': 'rule',
|
||||
'review-army-json-findings': 'rule',
|
||||
'review-army-red-team': 'behavior',
|
||||
'review-army-consensus': 'behavior',
|
||||
'review-army-simplification': 'behavior',
|
||||
'review-army-simplification-precision': 'behavior',
|
||||
'office-hours-spec-review': 'rule',
|
||||
'office-hours-brain-writeback': 'behavior',
|
||||
'gbrain-roundtrip-local': 'rule',
|
||||
'sync-gbrain-read-ready': 'rule',
|
||||
'sync-gbrain-read-unknown': 'rule',
|
||||
'office-hours-forcing-energy': 'behavior',
|
||||
'office-hours-builder-wildness': 'behavior',
|
||||
'plan-ceo-review': 'rule',
|
||||
'plan-ceo-review-selective': 'rule',
|
||||
'plan-ceo-review-benefits': 'rule',
|
||||
'plan-ceo-review-expansion-energy': 'behavior',
|
||||
'plan-eng-review': 'rule',
|
||||
'plan-eng-review-artifact': 'rule',
|
||||
'plan-eng-coverage-audit': 'rule',
|
||||
'plan-review-report': 'rule',
|
||||
'plan-ceo-review-plan-mode': 'rule',
|
||||
'plan-eng-review-plan-mode': 'rule',
|
||||
'plan-design-review-plan-mode': 'rule',
|
||||
'plan-devex-review-plan-mode': 'rule',
|
||||
'plan-mode-no-op': 'rule',
|
||||
'office-hours-auto-mode': 'rule',
|
||||
'auto-decide-preserved': 'rule',
|
||||
'auq-format-gate': 'rule',
|
||||
'plan-ceo-mode-routing': 'rule',
|
||||
'plan-design-with-ui-scope': 'rule',
|
||||
'tpa-present': 'rule',
|
||||
'tpa-absent-linux': 'rule',
|
||||
'tpa-broken': 'rule',
|
||||
'tpa-absent-darwin': 'rule',
|
||||
'tpa-apple-ban': 'rule',
|
||||
'ship-section-loading': 'rule',
|
||||
'plan-ceo-section-loading': 'rule',
|
||||
'carve-section-loading': 'rule',
|
||||
'plan-eng-finding-floor': 'rule',
|
||||
'plan-ceo-finding-floor': 'rule',
|
||||
'plan-design-finding-floor': 'rule',
|
||||
'plan-devex-finding-floor': 'rule',
|
||||
'plan-eng-multi-finding-batching': 'rule',
|
||||
'plan-ceo-split-overflow': 'rule',
|
||||
'setup-gbrain-remote': 'rule',
|
||||
'setup-gbrain-bad-token': 'rule',
|
||||
'setup-gbrain-path4-local-pglite': 'rule',
|
||||
'plan-ceo-review-format-mode': 'behavior',
|
||||
'plan-ceo-review-format-approach': 'behavior',
|
||||
'plan-eng-review-format-coverage': 'behavior',
|
||||
'plan-eng-review-format-kind': 'behavior',
|
||||
'office-hours-phase4-fork': 'behavior',
|
||||
'llm-judge-recommendation': 'judge',
|
||||
'plan-ceo-review-prosons-cadence': 'behavior',
|
||||
'plan-review-prosons-format': 'behavior',
|
||||
'plan-review-prosons-hardstop-neg': 'behavior',
|
||||
'plan-review-prosons-neutral-neg': 'behavior',
|
||||
'plan-tune-inspect': 'rule',
|
||||
'codex-offered-office-hours': 'rule',
|
||||
'codex-offered-ceo-review': 'rule',
|
||||
'codex-offered-design-review': 'rule',
|
||||
'codex-offered-eng-review': 'rule',
|
||||
'timeline-event-flow': 'rule',
|
||||
'context-recovery-artifacts': 'rule',
|
||||
'context-save-writes-file': 'rule',
|
||||
'context-restore-loads-latest': 'rule',
|
||||
'context-save-routing': 'rule',
|
||||
'context-save-then-restore-roundtrip': 'rule',
|
||||
'context-restore-fragment-match': 'rule',
|
||||
'context-restore-empty-state': 'rule',
|
||||
'context-restore-list-delegates': 'rule',
|
||||
'context-restore-legacy-compat': 'rule',
|
||||
'context-save-list-current-branch': 'rule',
|
||||
'context-save-list-all-branches': 'rule',
|
||||
'ship-base-branch': 'rule',
|
||||
'ship-local-workflow': 'rule',
|
||||
'ship-managed-hook-refresh': 'rule',
|
||||
'ship-unmanaged-hook-consent': 'rule',
|
||||
'ship-local-hook-preservation': 'rule',
|
||||
'ship-coverage-audit': 'rule',
|
||||
'ship-triage': 'rule',
|
||||
'ship-docsync-missing-marker': 'rule',
|
||||
'ship-docsync-missing-asset': 'rule',
|
||||
'ship-docsync-launch-failure': 'rule',
|
||||
'ship-docsync-timeout-unsettled': 'rule',
|
||||
'ship-docsync-late-result': 'rule',
|
||||
'ship-docsync-stale-before': 'rule',
|
||||
'ship-docsync-stale-after': 'rule',
|
||||
'ship-docsync-recovery': 'rule',
|
||||
'ship-docsync-completion': 'rule',
|
||||
'ship-docsync-current': 'rule',
|
||||
'ship-docsync-failure': 'rule',
|
||||
'ship-docsync-store': 'rule',
|
||||
'docsync-spawned': 'rule',
|
||||
'retro': 'rule',
|
||||
'retro-base-branch': 'rule',
|
||||
'cso-full-audit': 'rule',
|
||||
'cso-diff-mode': 'rule',
|
||||
'cso-infra-scope': 'rule',
|
||||
'learnings-show': 'rule',
|
||||
'document-release': 'rule',
|
||||
'codex-review': 'rule',
|
||||
'codex-discover-skill': 'rule',
|
||||
'codex-review-findings': 'rule',
|
||||
'outside-voice-codex-to-claude-code': 'rule',
|
||||
'outside-voice-claude-code-to-codex': 'rule',
|
||||
'outside-plan-disabled-no-fallback': 'rule',
|
||||
'codex-sol-scope-termination': 'rule',
|
||||
'design-consultation-core': 'rule',
|
||||
'design-consultation-existing': 'rule',
|
||||
'design-consultation-research': 'rule',
|
||||
'design-consultation-preview': 'rule',
|
||||
'plan-design-review-no-ui-scope': 'rule',
|
||||
'design-review-fix': 'rule',
|
||||
'design-review-detector-shim': 'rule',
|
||||
'design-review-detector-shim-dom': 'rule',
|
||||
'design-review-plugin-handoff': 'rule',
|
||||
'design-html-slop-gate': 'behavior',
|
||||
'diagram-triplet': 'rule',
|
||||
'diagram-authoring-quality': 'rule',
|
||||
'gstack-upgrade-happy-path': 'rule',
|
||||
'land-and-deploy-workflow': 'rule',
|
||||
'land-and-deploy-first-run': 'rule',
|
||||
'land-and-deploy-review-gate': 'rule',
|
||||
'canary-workflow': 'rule',
|
||||
'benchmark-workflow': 'rule',
|
||||
'setup-deploy-workflow': 'rule',
|
||||
'autoplan-dual-voice': 'rule',
|
||||
'benchmark-providers-live': 'rule',
|
||||
'scrape-match-path': 'behavior',
|
||||
'scrape-prototype-path': 'behavior',
|
||||
'skillify-happy-path': 'rule',
|
||||
'skillify-provenance-refusal': 'rule',
|
||||
'skillify-approval-reject': 'rule',
|
||||
'journey-ideation': 'rule',
|
||||
'journey-plan-eng': 'rule',
|
||||
'journey-debug': 'rule',
|
||||
'journey-qa': 'rule',
|
||||
'journey-code-review': 'rule',
|
||||
'journey-ship': 'rule',
|
||||
'journey-docs': 'rule',
|
||||
'journey-retro': 'rule',
|
||||
'journey-design-system': 'rule',
|
||||
'journey-visual-qa': 'rule',
|
||||
'ios-qa-device': 'rule',
|
||||
'arm-benchmark-native-overbuild': 'rule',
|
||||
'arm-benchmark-crud-endpoint': 'rule',
|
||||
'arm-benchmark-bugfix-decoys': 'rule',
|
||||
'office-hours-section-loading': 'rule',
|
||||
'office-hours-design-draft': 'rule',
|
||||
'plan-decision-classification': 'rule',
|
||||
'plan-devex-peer-comparison-classification': 'rule',
|
||||
'health-reporting': 'rule',
|
||||
'overlay-harness-claude-dedicated-tools-vs-bash': 'rule',
|
||||
'overlay-harness-opus-4-7-effort-match-trivial': 'rule',
|
||||
'overlay-harness-opus-4-7-literal-interpretation': 'rule',
|
||||
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule',
|
||||
'journey-negatives': 'rule',
|
||||
'review/SKILL.md workflow': 'judge',
|
||||
'setup-browser-cookies/SKILL.md workflow': 'judge',
|
||||
'browse/SKILL.md reference': 'judge',
|
||||
'setup block': 'judge',
|
||||
'qa/SKILL.md workflow': 'judge',
|
||||
'qa/SKILL.md health rubric': 'judge',
|
||||
'qa/SKILL.md anti-refusal': 'judge',
|
||||
'cross-skill greptile consistency': 'judge',
|
||||
'ship/SKILL.md workflow': 'judge',
|
||||
'document-release/SKILL.md workflow': 'judge',
|
||||
'plan-ceo-review/SKILL.md modes': 'judge',
|
||||
'plan-eng-review/SKILL.md sections': 'judge',
|
||||
'plan-design-review/SKILL.md passes': 'judge',
|
||||
'design-review/SKILL.md fix loop': 'judge',
|
||||
'design-consultation/SKILL.md research': 'judge',
|
||||
'land-and-deploy/SKILL.md workflow': 'judge',
|
||||
'canary/SKILL.md monitoring loop': 'judge',
|
||||
'benchmark/SKILL.md perf collection': 'judge',
|
||||
'setup-deploy/SKILL.md platform setup': 'judge',
|
||||
'retro/SKILL.md instructions': 'judge',
|
||||
'qa-only/SKILL.md workflow': 'judge',
|
||||
'gstack-upgrade/SKILL.md upgrade flow': 'judge',
|
||||
'sync-gbrain/SKILL.md read-only readiness': 'judge',
|
||||
'voice directive tone': 'judge',
|
||||
};
|
||||
|
||||
/**
|
||||
* One-line tolerance for every behavior-kind case: why an occasional
|
||||
* deviation is acceptable product behavior. Keys equal the behavior ids of
|
||||
* E2E_KINDS; values are non-empty.
|
||||
*/
|
||||
export const BEHAVIOR_WHY: Record<string, string> = {
|
||||
'shared-libs-opportunity-judgment':
|
||||
"Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.",
|
||||
'review-design-lite':
|
||||
"How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.",
|
||||
'review-army-red-team':
|
||||
"Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.",
|
||||
'review-army-consensus':
|
||||
"Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.",
|
||||
'review-army-simplification':
|
||||
"Flagging the planted unnecessary structure is an advisory-lens quality judgment.",
|
||||
'review-army-simplification-precision':
|
||||
"Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.",
|
||||
'office-hours-forcing-energy':
|
||||
"The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.",
|
||||
'office-hours-builder-wildness':
|
||||
"Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.",
|
||||
'office-hours-brain-writeback':
|
||||
"The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.",
|
||||
'office-hours-phase4-fork':
|
||||
"Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.",
|
||||
'plan-ceo-review-expansion-energy':
|
||||
"Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.",
|
||||
'plan-ceo-review-format-mode':
|
||||
"Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.",
|
||||
'plan-ceo-review-format-approach':
|
||||
"Approach-menu Completeness wording is live formatting of one AskUserQuestion.",
|
||||
'plan-eng-review-format-coverage':
|
||||
"Coverage-issue Completeness wording is live formatting of one AskUserQuestion.",
|
||||
'plan-eng-review-format-kind':
|
||||
"Kind-note wording is live formatting of one AskUserQuestion.",
|
||||
'plan-ceo-review-prosons-cadence':
|
||||
"Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.",
|
||||
'plan-review-prosons-format':
|
||||
"The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.",
|
||||
'plan-review-prosons-hardstop-neg':
|
||||
"Not using the hard-stop escape on an ordinary decision is live formatting of one question.",
|
||||
'plan-review-prosons-neutral-neg':
|
||||
"Avoiding neutral posture and naming a because-reason is live formatting of one question.",
|
||||
'design-html-slop-gate':
|
||||
"How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.",
|
||||
'scrape-match-path':
|
||||
"The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.",
|
||||
'scrape-prototype-path':
|
||||
"The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.",
|
||||
};
|
||||
@@ -4,7 +4,7 @@ import * as path from 'node:path';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
|
||||
import { JUDGE_MS } from './eval-budgets';
|
||||
import type { JudgeScore } from './llm-judge';
|
||||
import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge';
|
||||
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input';
|
||||
import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache,
|
||||
type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache';
|
||||
@@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres
|
||||
|| typeof value.reasoning !== 'string'
|
||||
|| (structuredResponse && (!value.reasoning.trim()
|
||||
|| value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false;
|
||||
return (['clarity', 'completeness', 'actionability'] as const).every(key =>
|
||||
return JUDGE_SCORE_DIMENSIONS.every(key =>
|
||||
typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5);
|
||||
}
|
||||
|
||||
const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 };
|
||||
|
||||
/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */
|
||||
export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array<JudgeScore & EvalCacheValue> } {
|
||||
if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples'
|
||||
|| !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES
|
||||
|| !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false;
|
||||
const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS);
|
||||
return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]);
|
||||
}
|
||||
|
||||
export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null;
|
||||
lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null;
|
||||
/** The attempt guard is rechecked after synchronous input/provenance reads. */
|
||||
publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined;
|
||||
publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined;
|
||||
} {
|
||||
const env = opts.env ?? process.env;
|
||||
const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined };
|
||||
const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined };
|
||||
const pr = Number(env.EVALS_CACHE_PR);
|
||||
// Runtime ID is the immutable CI image manifest, not a mutable image tag.
|
||||
// Nonstandard Node/Bun preload code or custom model endpoints need a separate
|
||||
@@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
|
||||
prompts: { [opts.testName]: prompt },
|
||||
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
|
||||
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1,
|
||||
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0,
|
||||
panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' },
|
||||
...(opts.stream ? { stream: true } : {}),
|
||||
...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
|
||||
response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) },
|
||||
@@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
return {
|
||||
lookup() {
|
||||
const result = lookupEvalInputCache({ ...common, identity: before,
|
||||
validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) });
|
||||
validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) });
|
||||
return result.status === 'reused'
|
||||
? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null;
|
||||
? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null;
|
||||
},
|
||||
publish(scores, isActive = () => true) {
|
||||
publish(samples, isActive = () => true) {
|
||||
// Caller reaches here ONLY after its actual assertions passed. A later
|
||||
// failed case in the file does not erase this independently completed case.
|
||||
if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
|
||||
const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) };
|
||||
if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
|
||||
const after = currentIdentity();
|
||||
const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID;
|
||||
if (!after || !runId || !isActive()) return;
|
||||
@@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
cancelled: false, skipped: 0, failed: 0, passed: 1,
|
||||
cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }],
|
||||
source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() },
|
||||
result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning },
|
||||
result: panel,
|
||||
} });
|
||||
// A slow synchronous write can consume the recording allowance. The
|
||||
// caller withdraws this new receipt if its final deadline check fails.
|
||||
|
||||
Reference in new issue
Block a user