test(evals): classify every live case and re-select a case when its kind changes

E2E_KINDS: rule by default (191 E2E ids), 22 behavior cases whose verdict is
a live model choice with an acceptable sub-100% per-trial rate, each with a
BEHAVIOR_WHY tolerance, and 25 judge entries (the 24 workflow judges plus the
fixed-fixture llm-judge-recommendation rubric check). Contract-shaped cases
(ask-before-decide, plan-mode no-writes, mandated steps, secrets, the batching
floor) stay rule. Behavior requires a known literal registration and an exact
Bun test name so the case runs as its own trial shard.

Map-diff selection now diffs E2E_KINDS and BEHAVIOR_WHY per key, and a base
revision without them selects every key, so a kind flip runs the panel it
introduces. test/eval-kinds.test.ts enforces coverage, tolerances,
isolatability and the reviewed counts, printing the literal to add.
This commit is contained in:
garrytan committed 2026-09-29 19:11:04 +00:00
1 parent 3f68572cab
commit 0292ee3f6f
3 files changed
+216 -51

No files matched your search

+107
View File
@@ -0,0 +1,107 @@
/**
* Eval kind registry (E2E_KINDS / BEHAVIOR_WHY in touchfiles-data.ts). The
* kind fixes a case's trial policy before the run, so the registry must cover
* every live case exactly once, every behavior case must name its tolerated
* deviation, and a behavior case must be isolatable as its own trial shard.
* A kind edit must re-select the case in the PR lane (map-diff).
*/
import { describe, expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { BEHAVIOR_WHY, E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data';
import { diffTouchfileMapsCore, type TouchfileMaps } from './helpers/test-selection';
import { CASE_TEST_NAMES, fileCaseRegistration } from '../scripts/test-paid-shards';
import { isPaidTestFile } from './helpers/paid-test-set';
const ROOT = path.resolve(import.meta.dir, '..');
const KIND_RULE = "Pick the kind by what can make the verdict differ between two runs of the same commit: 'rule' when nothing "
+ "stochastic decides it or it checks a contract the product must meet every run (the default); 'behavior' when a live "
+ "model choice decides it and a sub-100% per-trial rate is acceptable (add a BEHAVIOR_WHY line); 'judge' when the only "
+ 'stochastic step is an LLM judge scoring a fixed input.';
const liveIds = [...Object.keys(E2E_TIERS), ...Object.keys(LLM_JUDGE_TOUCHFILES)];
const behaviorIds = Object.keys(E2E_KINDS).filter(id => E2E_KINDS[id] === 'behavior').sort();
describe('E2E_KINDS registry', () => {
test('every live case has exactly one kind and no kind names a dead case', () => {
const missing = liveIds.filter(id => !(id in E2E_KINDS));
expect(missing.length, missing.length ? `add to E2E_KINDS:\n${missing.map(id => ` '${id}': 'rule', // <reason>`).join('\n')}\n${KIND_RULE}` : '').toBe(0);
const unknown = Object.keys(E2E_KINDS).filter(id => !liveIds.includes(id));
expect(unknown, `E2E_KINDS names ids that are neither E2E_TIERS nor LLM_JUDGE_TOUCHFILES keys`).toEqual([]);
expect(new Set(liveIds).size).toBe(liveIds.length);
});
test('kinds are rule, behavior or judge; every LLM-judge entry is judge-kind', () => {
for (const [id, kind] of Object.entries(E2E_KINDS)) expect(['rule', 'behavior', 'judge'], id).toContain(kind);
for (const id of Object.keys(LLM_JUDGE_TOUCHFILES)) expect(E2E_KINDS[id], `${id}: a workflow judge scores a fixed input`).toBe('judge');
});
test('BEHAVIOR_WHY names the tolerance of exactly the behavior cases', () => {
expect(Object.keys(BEHAVIOR_WHY).sort()).toEqual(behaviorIds);
for (const id of behaviorIds) {
expect(BEHAVIOR_WHY[id]!.trim().length, `${id}: BEHAVIOR_WHY must say why an occasional deviation is acceptable`).toBeGreaterThanOrEqual(30);
}
});
test('a behavior case is an isolatable trial shard: known literal registration and an exact Bun test name', () => {
for (const id of behaviorIds) {
const files = E2E_TOUCHFILES[id]!.filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file));
expect(files.length, `${id}: no paid test file registers it`).toBeGreaterThan(0);
for (const file of files) {
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
expect(fileCaseRegistration(file, source).known, `${id}: ${file} has a computed registration; behavior needs a literal one`).toBe(true);
const name = CASE_TEST_NAMES[id] ?? id;
const literal = new RegExp(`\\b(?:test(?:\\.serial|\\.concurrent)?|testIfSelected|testConcurrentIfSelected)\\(\\s*(['"\`])${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\1`);
expect(literal.test(source), `${id}: ${file} must register the Bun test named '${name}'`).toBe(true);
}
}
});
test('the classification is the reviewed one: rule by default, 22 behavior, 25 judge', () => {
const counts = Object.values(E2E_KINDS).reduce<Record<string, number>>((acc, kind) => ({ ...acc, [kind]: (acc[kind] ?? 0) + 1 }), {});
expect(counts).toEqual({ rule: liveIds.length - 22 - 25, behavior: 22, judge: 25 });
// Contract-shaped cases stay rule: ask-before-decide, plan-mode no-writes,
// mandated steps, secrets, and the batching floor never ride a majority.
for (const id of ['plan-ceo-mode-routing', 'plan-eng-multi-finding-batching', 'plan-design-review-plan-mode',
'plan-eng-review-plan-mode', 'plan-ceo-section-loading', 'setup-gbrain-bad-token', 'qa-only-no-fix', 'review-sql-injection']) {
expect(E2E_KINDS[id], id).toBe('rule');
}
});
});
describe('kind edits re-select their case (map-diff)', () => {
const base = (): TouchfileMaps => ({
E2E_TOUCHFILES: { alpha: ['a/**'], beta: ['b/**'] },
E2E_TIERS: { alpha: 'gate', beta: 'periodic' },
LLM_JUDGE_TOUCHFILES: { 'judge one': ['j/SKILL.md'] },
GLOBAL_TOUCHFILES: [],
E2E_KINDS: { alpha: 'rule', beta: 'rule', 'judge one': 'judge' },
BEHAVIOR_WHY: {},
});
test('a rule -> behavior flip selects exactly that case', () => {
const next = base();
next.E2E_KINDS = { ...next.E2E_KINDS, beta: 'behavior' };
next.BEHAVIOR_WHY = { beta: 'tolerated deviation' };
expect(diffTouchfileMapsCore(base(), next).changedTests).toEqual(['beta']);
});
test('a BEHAVIOR_WHY edit alone selects its case', () => {
const old = base(); old.E2E_KINDS!.beta = 'behavior'; old.BEHAVIOR_WHY = { beta: 'one' };
const next = base(); next.E2E_KINDS!.beta = 'behavior'; next.BEHAVIOR_WHY = { beta: 'two' };
expect(diffTouchfileMapsCore(old, next).changedTests).toEqual(['beta']);
});
test('a base revision without the kind maps selects every key', () => {
const old = base(); delete old.E2E_KINDS; delete old.BEHAVIOR_WHY;
expect(diffTouchfileMapsCore(old, base()).changedTests).toEqual(['alpha', 'beta', 'judge one']);
});
test('dropping a kind entry while the case lives on counts as changed, not removed', () => {
const next = base(); delete next.E2E_KINDS!.alpha;
const result = diffTouchfileMapsCore(base(), next);
expect(result.changedTests).toEqual(['alpha']);
expect(result.removedTests).toEqual([]);
});
});
+16 -3
View File
@@ -34,6 +34,8 @@ import {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
} from './touchfiles-data';
/** Repo-relative path of the pure-data file (the map-diff subject). */
@@ -145,6 +147,9 @@ export interface TouchfileMaps {
E2E_TIERS: Record<string, string>;
LLM_JUDGE_TOUCHFILES: Record<string, string[]>;
GLOBAL_TOUCHFILES: string[];
/** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */
E2E_KINDS?: Record<string, string>;
BEHAVIOR_WHY?: Record<string, string>;
}
export type MapDiffCause =
@@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
};
function isStringArray(v: unknown): v is string[] {
@@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps {
return isRecordOfStringArrays(o.E2E_TOUCHFILES)
&& isRecordOfStrings(o.E2E_TIERS)
&& isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES)
&& isStringArray(o.GLOBAL_TOUCHFILES);
&& isStringArray(o.GLOBAL_TOUCHFILES)
&& (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS))
&& (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY));
}
/**
* Pure map-diff core (injectable for tests — no git, no filesystem).
*
* A key counts as CHANGED when it was added to any per-key map, its dep-list
* array differs, or its tier value flipped. A key counts as REMOVED only when
* array differs, or its tier, kind or behavior tolerance changed. A per-key
* map missing on the old side (a base revision older than E2E_KINDS /
* BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when
* it is gone from every new per-key map; a key dropped from one map but still
* present in another (e.g. tier entry deleted, touchfile entry kept) counts
* as changed — conservative, because the test still exists with a different
@@ -212,7 +223,7 @@ export function diffTouchfileMapsCore(
oldMaps: TouchfileMaps,
newMaps: TouchfileMaps,
): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } {
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const;
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const;
const changed = new Set<string>();
const rawRemoved = new Set<string>();
@@ -290,6 +301,8 @@ export function diffTouchfileMaps(
' E2E_TIERS: m.E2E_TIERS,',
' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,',
' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,',
' E2E_KINDS: m.E2E_KINDS,',
' BEHAVIOR_WHY: m.BEHAVIOR_WHY,',
'}));',
'',
].join('\n'));
+93 -48
View File
@@ -1597,7 +1597,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'shared-libs-unsupported-git': 'rule',
'shared-libs-review-lifecycle': 'rule',
'shared-libs-review-revalidation': 'rule',
'shared-libs-opportunity-judgment': 'rule',
'shared-libs-opportunity-judgment': 'behavior',
'shared-libs-pr-coverage': 'rule',
'shared-libs-plan-callers': 'rule',
'browse-basic': 'rule',
@@ -1634,7 +1634,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'review-sql-injection': 'rule',
'review-enum-completeness': 'rule',
'review-base-branch': 'rule',
'review-design-lite': 'rule',
'review-design-lite': 'behavior',
'review-coverage-audit': 'rule',
'review-dashboard-via': 'rule',
'review-army-migration-safety': 'rule',
@@ -1642,21 +1642,21 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'review-army-delivery-audit': 'rule',
'review-army-quality-score': 'rule',
'review-army-json-findings': 'rule',
'review-army-red-team': 'rule',
'review-army-consensus': 'rule',
'review-army-simplification': 'rule',
'review-army-simplification-precision': 'rule',
'review-army-red-team': 'behavior',
'review-army-consensus': 'behavior',
'review-army-simplification': 'behavior',
'review-army-simplification-precision': 'behavior',
'office-hours-spec-review': 'rule',
'office-hours-brain-writeback': 'rule',
'office-hours-brain-writeback': 'behavior',
'gbrain-roundtrip-local': 'rule',
'sync-gbrain-read-ready': 'rule',
'sync-gbrain-read-unknown': 'rule',
'office-hours-forcing-energy': 'rule',
'office-hours-builder-wildness': 'rule',
'office-hours-forcing-energy': 'behavior',
'office-hours-builder-wildness': 'behavior',
'plan-ceo-review': 'rule',
'plan-ceo-review-selective': 'rule',
'plan-ceo-review-benefits': 'rule',
'plan-ceo-review-expansion-energy': 'rule',
'plan-ceo-review-expansion-energy': 'behavior',
'plan-eng-review': 'rule',
'plan-eng-review-artifact': 'rule',
'plan-eng-coverage-audit': 'rule',
@@ -1688,16 +1688,16 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'setup-gbrain-remote': 'rule',
'setup-gbrain-bad-token': 'rule',
'setup-gbrain-path4-local-pglite': 'rule',
'plan-ceo-review-format-mode': 'rule',
'plan-ceo-review-format-approach': 'rule',
'plan-eng-review-format-coverage': 'rule',
'plan-eng-review-format-kind': 'rule',
'office-hours-phase4-fork': 'rule',
'llm-judge-recommendation': 'rule',
'plan-ceo-review-prosons-cadence': 'rule',
'plan-review-prosons-format': 'rule',
'plan-review-prosons-hardstop-neg': 'rule',
'plan-review-prosons-neutral-neg': 'rule',
'plan-ceo-review-format-mode': 'behavior',
'plan-ceo-review-format-approach': 'behavior',
'plan-eng-review-format-coverage': 'behavior',
'plan-eng-review-format-kind': 'behavior',
'office-hours-phase4-fork': 'behavior',
'llm-judge-recommendation': 'judge',
'plan-ceo-review-prosons-cadence': 'behavior',
'plan-review-prosons-format': 'behavior',
'plan-review-prosons-hardstop-neg': 'behavior',
'plan-review-prosons-neutral-neg': 'behavior',
'plan-tune-inspect': 'rule',
'codex-offered-office-hours': 'rule',
'codex-offered-ceo-review': 'rule',
@@ -1758,7 +1758,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'design-review-detector-shim': 'rule',
'design-review-detector-shim-dom': 'rule',
'design-review-plugin-handoff': 'rule',
'design-html-slop-gate': 'rule',
'design-html-slop-gate': 'behavior',
'diagram-triplet': 'rule',
'diagram-authoring-quality': 'rule',
'gstack-upgrade-happy-path': 'rule',
@@ -1770,8 +1770,8 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'setup-deploy-workflow': 'rule',
'autoplan-dual-voice': 'rule',
'benchmark-providers-live': 'rule',
'scrape-match-path': 'rule',
'scrape-prototype-path': 'rule',
'scrape-match-path': 'behavior',
'scrape-prototype-path': 'behavior',
'skillify-happy-path': 'rule',
'skillify-provenance-refusal': 'rule',
'skillify-approval-reject': 'rule',
@@ -1799,30 +1799,30 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'overlay-harness-opus-4-7-literal-interpretation': 'rule',
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule',
'journey-negatives': 'rule',
'review/SKILL.md workflow': 'rule',
'setup-browser-cookies/SKILL.md workflow': 'rule',
'browse/SKILL.md reference': 'rule',
'setup block': 'rule',
'qa/SKILL.md workflow': 'rule',
'qa/SKILL.md health rubric': 'rule',
'qa/SKILL.md anti-refusal': 'rule',
'cross-skill greptile consistency': 'rule',
'ship/SKILL.md workflow': 'rule',
'document-release/SKILL.md workflow': 'rule',
'plan-ceo-review/SKILL.md modes': 'rule',
'plan-eng-review/SKILL.md sections': 'rule',
'plan-design-review/SKILL.md passes': 'rule',
'design-review/SKILL.md fix loop': 'rule',
'design-consultation/SKILL.md research': 'rule',
'land-and-deploy/SKILL.md workflow': 'rule',
'canary/SKILL.md monitoring loop': 'rule',
'benchmark/SKILL.md perf collection': 'rule',
'setup-deploy/SKILL.md platform setup': 'rule',
'retro/SKILL.md instructions': 'rule',
'qa-only/SKILL.md workflow': 'rule',
'gstack-upgrade/SKILL.md upgrade flow': 'rule',
'sync-gbrain/SKILL.md read-only readiness': 'rule',
'voice directive tone': 'rule',
'review/SKILL.md workflow': 'judge',
'setup-browser-cookies/SKILL.md workflow': 'judge',
'browse/SKILL.md reference': 'judge',
'setup block': 'judge',
'qa/SKILL.md workflow': 'judge',
'qa/SKILL.md health rubric': 'judge',
'qa/SKILL.md anti-refusal': 'judge',
'cross-skill greptile consistency': 'judge',
'ship/SKILL.md workflow': 'judge',
'document-release/SKILL.md workflow': 'judge',
'plan-ceo-review/SKILL.md modes': 'judge',
'plan-eng-review/SKILL.md sections': 'judge',
'plan-design-review/SKILL.md passes': 'judge',
'design-review/SKILL.md fix loop': 'judge',
'design-consultation/SKILL.md research': 'judge',
'land-and-deploy/SKILL.md workflow': 'judge',
'canary/SKILL.md monitoring loop': 'judge',
'benchmark/SKILL.md perf collection': 'judge',
'setup-deploy/SKILL.md platform setup': 'judge',
'retro/SKILL.md instructions': 'judge',
'qa-only/SKILL.md workflow': 'judge',
'gstack-upgrade/SKILL.md upgrade flow': 'judge',
'sync-gbrain/SKILL.md read-only readiness': 'judge',
'voice directive tone': 'judge',
};
/**
@@ -1830,4 +1830,49 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
* deviation is acceptable product behavior. Keys equal the behavior ids of
* E2E_KINDS; values are non-empty.
*/
export const BEHAVIOR_WHY: Record<string, string> = {};
export const BEHAVIOR_WHY: Record<string, string> = {
'shared-libs-opportunity-judgment':
"Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.",
'review-design-lite':
"How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.",
'review-army-red-team':
"Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.",
'review-army-consensus':
"Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.",
'review-army-simplification':
"Flagging the planted unnecessary structure is an advisory-lens quality judgment.",
'review-army-simplification-precision':
"Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.",
'office-hours-forcing-energy':
"The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.",
'office-hours-builder-wildness':
"Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.",
'office-hours-brain-writeback':
"The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.",
'office-hours-phase4-fork':
"Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.",
'plan-ceo-review-expansion-energy':
"Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.",
'plan-ceo-review-format-mode':
"Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.",
'plan-ceo-review-format-approach':
"Approach-menu Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-coverage':
"Coverage-issue Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-kind':
"Kind-note wording is live formatting of one AskUserQuestion.",
'plan-ceo-review-prosons-cadence':
"Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.",
'plan-review-prosons-format':
"The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.",
'plan-review-prosons-hardstop-neg':
"Not using the hard-stop escape on an ordinary decision is live formatting of one question.",
'plan-review-prosons-neutral-neg':
"Avoiding neutral posture and naming a because-reason is live formatting of one question.",
'design-html-slop-gate':
"How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.",
'scrape-match-path':
"The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.",
'scrape-prototype-path':
"The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.",
};