test(evals): classify every live case and re-select a case when its kind changes

E2E_KINDS: rule by default (191 E2E ids), 22 behavior cases whose verdict is
a live model choice with an acceptable sub-100% per-trial rate, each with a
BEHAVIOR_WHY tolerance, and 25 judge entries (the 24 workflow judges plus the
fixed-fixture llm-judge-recommendation rubric check). Contract-shaped cases
(ask-before-decide, plan-mode no-writes, mandated steps, secrets, the batching
floor) stay rule. Behavior requires a known literal registration and an exact
Bun test name so the case runs as its own trial shard.

Map-diff selection now diffs E2E_KINDS and BEHAVIOR_WHY per key, and a base
revision without them selects every key, so a kind flip runs the panel it
introduces. test/eval-kinds.test.ts enforces coverage, tolerances,
isolatability and the reviewed counts, printing the literal to add.
This commit is contained in:
garrytan committed 2026-09-29 19:11:04 +00:00
1 parent 3f68572cab
commit 0292ee3f6f
3 files changed
+216 -51

No files matched your search

+16 -3
View File
@@ -34,6 +34,8 @@ import {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
} from './touchfiles-data';
/** Repo-relative path of the pure-data file (the map-diff subject). */
@@ -145,6 +147,9 @@ export interface TouchfileMaps {
E2E_TIERS: Record<string, string>;
LLM_JUDGE_TOUCHFILES: Record<string, string[]>;
GLOBAL_TOUCHFILES: string[];
/** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */
E2E_KINDS?: Record<string, string>;
BEHAVIOR_WHY?: Record<string, string>;
}
export type MapDiffCause =
@@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = {
E2E_TIERS,
LLM_JUDGE_TOUCHFILES,
GLOBAL_TOUCHFILES,
E2E_KINDS,
BEHAVIOR_WHY,
};
function isStringArray(v: unknown): v is string[] {
@@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps {
return isRecordOfStringArrays(o.E2E_TOUCHFILES)
&& isRecordOfStrings(o.E2E_TIERS)
&& isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES)
&& isStringArray(o.GLOBAL_TOUCHFILES);
&& isStringArray(o.GLOBAL_TOUCHFILES)
&& (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS))
&& (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY));
}
/**
* Pure map-diff core (injectable for tests — no git, no filesystem).
*
* A key counts as CHANGED when it was added to any per-key map, its dep-list
* array differs, or its tier value flipped. A key counts as REMOVED only when
* array differs, or its tier, kind or behavior tolerance changed. A per-key
* map missing on the old side (a base revision older than E2E_KINDS /
* BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when
* it is gone from every new per-key map; a key dropped from one map but still
* present in another (e.g. tier entry deleted, touchfile entry kept) counts
* as changed — conservative, because the test still exists with a different
@@ -212,7 +223,7 @@ export function diffTouchfileMapsCore(
oldMaps: TouchfileMaps,
newMaps: TouchfileMaps,
): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } {
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const;
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const;
const changed = new Set<string>();
const rawRemoved = new Set<string>();
@@ -290,6 +301,8 @@ export function diffTouchfileMaps(
' E2E_TIERS: m.E2E_TIERS,',
' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,',
' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,',
' E2E_KINDS: m.E2E_KINDS,',
' BEHAVIOR_WHY: m.BEHAVIOR_WHY,',
'}));',
'',
].join('\n'));
+93 -48
View File
@@ -1597,7 +1597,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'shared-libs-unsupported-git': 'rule',
'shared-libs-review-lifecycle': 'rule',
'shared-libs-review-revalidation': 'rule',
'shared-libs-opportunity-judgment': 'rule',
'shared-libs-opportunity-judgment': 'behavior',
'shared-libs-pr-coverage': 'rule',
'shared-libs-plan-callers': 'rule',
'browse-basic': 'rule',
@@ -1634,7 +1634,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'review-sql-injection': 'rule',
'review-enum-completeness': 'rule',
'review-base-branch': 'rule',
'review-design-lite': 'rule',
'review-design-lite': 'behavior',
'review-coverage-audit': 'rule',
'review-dashboard-via': 'rule',
'review-army-migration-safety': 'rule',
@@ -1642,21 +1642,21 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'review-army-delivery-audit': 'rule',
'review-army-quality-score': 'rule',
'review-army-json-findings': 'rule',
'review-army-red-team': 'rule',
'review-army-consensus': 'rule',
'review-army-simplification': 'rule',
'review-army-simplification-precision': 'rule',
'review-army-red-team': 'behavior',
'review-army-consensus': 'behavior',
'review-army-simplification': 'behavior',
'review-army-simplification-precision': 'behavior',
'office-hours-spec-review': 'rule',
'office-hours-brain-writeback': 'rule',
'office-hours-brain-writeback': 'behavior',
'gbrain-roundtrip-local': 'rule',
'sync-gbrain-read-ready': 'rule',
'sync-gbrain-read-unknown': 'rule',
'office-hours-forcing-energy': 'rule',
'office-hours-builder-wildness': 'rule',
'office-hours-forcing-energy': 'behavior',
'office-hours-builder-wildness': 'behavior',
'plan-ceo-review': 'rule',
'plan-ceo-review-selective': 'rule',
'plan-ceo-review-benefits': 'rule',
'plan-ceo-review-expansion-energy': 'rule',
'plan-ceo-review-expansion-energy': 'behavior',
'plan-eng-review': 'rule',
'plan-eng-review-artifact': 'rule',
'plan-eng-coverage-audit': 'rule',
@@ -1688,16 +1688,16 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'setup-gbrain-remote': 'rule',
'setup-gbrain-bad-token': 'rule',
'setup-gbrain-path4-local-pglite': 'rule',
'plan-ceo-review-format-mode': 'rule',
'plan-ceo-review-format-approach': 'rule',
'plan-eng-review-format-coverage': 'rule',
'plan-eng-review-format-kind': 'rule',
'office-hours-phase4-fork': 'rule',
'llm-judge-recommendation': 'rule',
'plan-ceo-review-prosons-cadence': 'rule',
'plan-review-prosons-format': 'rule',
'plan-review-prosons-hardstop-neg': 'rule',
'plan-review-prosons-neutral-neg': 'rule',
'plan-ceo-review-format-mode': 'behavior',
'plan-ceo-review-format-approach': 'behavior',
'plan-eng-review-format-coverage': 'behavior',
'plan-eng-review-format-kind': 'behavior',
'office-hours-phase4-fork': 'behavior',
'llm-judge-recommendation': 'judge',
'plan-ceo-review-prosons-cadence': 'behavior',
'plan-review-prosons-format': 'behavior',
'plan-review-prosons-hardstop-neg': 'behavior',
'plan-review-prosons-neutral-neg': 'behavior',
'plan-tune-inspect': 'rule',
'codex-offered-office-hours': 'rule',
'codex-offered-ceo-review': 'rule',
@@ -1758,7 +1758,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'design-review-detector-shim': 'rule',
'design-review-detector-shim-dom': 'rule',
'design-review-plugin-handoff': 'rule',
'design-html-slop-gate': 'rule',
'design-html-slop-gate': 'behavior',
'diagram-triplet': 'rule',
'diagram-authoring-quality': 'rule',
'gstack-upgrade-happy-path': 'rule',
@@ -1770,8 +1770,8 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'setup-deploy-workflow': 'rule',
'autoplan-dual-voice': 'rule',
'benchmark-providers-live': 'rule',
'scrape-match-path': 'rule',
'scrape-prototype-path': 'rule',
'scrape-match-path': 'behavior',
'scrape-prototype-path': 'behavior',
'skillify-happy-path': 'rule',
'skillify-provenance-refusal': 'rule',
'skillify-approval-reject': 'rule',
@@ -1799,30 +1799,30 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
'overlay-harness-opus-4-7-literal-interpretation': 'rule',
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule',
'journey-negatives': 'rule',
'review/SKILL.md workflow': 'rule',
'setup-browser-cookies/SKILL.md workflow': 'rule',
'browse/SKILL.md reference': 'rule',
'setup block': 'rule',
'qa/SKILL.md workflow': 'rule',
'qa/SKILL.md health rubric': 'rule',
'qa/SKILL.md anti-refusal': 'rule',
'cross-skill greptile consistency': 'rule',
'ship/SKILL.md workflow': 'rule',
'document-release/SKILL.md workflow': 'rule',
'plan-ceo-review/SKILL.md modes': 'rule',
'plan-eng-review/SKILL.md sections': 'rule',
'plan-design-review/SKILL.md passes': 'rule',
'design-review/SKILL.md fix loop': 'rule',
'design-consultation/SKILL.md research': 'rule',
'land-and-deploy/SKILL.md workflow': 'rule',
'canary/SKILL.md monitoring loop': 'rule',
'benchmark/SKILL.md perf collection': 'rule',
'setup-deploy/SKILL.md platform setup': 'rule',
'retro/SKILL.md instructions': 'rule',
'qa-only/SKILL.md workflow': 'rule',
'gstack-upgrade/SKILL.md upgrade flow': 'rule',
'sync-gbrain/SKILL.md read-only readiness': 'rule',
'voice directive tone': 'rule',
'review/SKILL.md workflow': 'judge',
'setup-browser-cookies/SKILL.md workflow': 'judge',
'browse/SKILL.md reference': 'judge',
'setup block': 'judge',
'qa/SKILL.md workflow': 'judge',
'qa/SKILL.md health rubric': 'judge',
'qa/SKILL.md anti-refusal': 'judge',
'cross-skill greptile consistency': 'judge',
'ship/SKILL.md workflow': 'judge',
'document-release/SKILL.md workflow': 'judge',
'plan-ceo-review/SKILL.md modes': 'judge',
'plan-eng-review/SKILL.md sections': 'judge',
'plan-design-review/SKILL.md passes': 'judge',
'design-review/SKILL.md fix loop': 'judge',
'design-consultation/SKILL.md research': 'judge',
'land-and-deploy/SKILL.md workflow': 'judge',
'canary/SKILL.md monitoring loop': 'judge',
'benchmark/SKILL.md perf collection': 'judge',
'setup-deploy/SKILL.md platform setup': 'judge',
'retro/SKILL.md instructions': 'judge',
'qa-only/SKILL.md workflow': 'judge',
'gstack-upgrade/SKILL.md upgrade flow': 'judge',
'sync-gbrain/SKILL.md read-only readiness': 'judge',
'voice directive tone': 'judge',
};
/**
@@ -1830,4 +1830,49 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
* deviation is acceptable product behavior. Keys equal the behavior ids of
* E2E_KINDS; values are non-empty.
*/
export const BEHAVIOR_WHY: Record<string, string> = {};
export const BEHAVIOR_WHY: Record<string, string> = {
'shared-libs-opportunity-judgment':
"Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.",
'review-design-lite':
"How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.",
'review-army-red-team':
"Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.",
'review-army-consensus':
"Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.",
'review-army-simplification':
"Flagging the planted unnecessary structure is an advisory-lens quality judgment.",
'review-army-simplification-precision':
"Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.",
'office-hours-forcing-energy':
"The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.",
'office-hours-builder-wildness':
"Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.",
'office-hours-brain-writeback':
"The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.",
'office-hours-phase4-fork':
"Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.",
'plan-ceo-review-expansion-energy':
"Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.",
'plan-ceo-review-format-mode':
"Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.",
'plan-ceo-review-format-approach':
"Approach-menu Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-coverage':
"Coverage-issue Completeness wording is live formatting of one AskUserQuestion.",
'plan-eng-review-format-kind':
"Kind-note wording is live formatting of one AskUserQuestion.",
'plan-ceo-review-prosons-cadence':
"Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.",
'plan-review-prosons-format':
"The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.",
'plan-review-prosons-hardstop-neg':
"Not using the hard-stop escape on an ordinary decision is live formatting of one question.",
'plan-review-prosons-neutral-neg':
"Avoiding neutral posture and naming a because-reason is live formatting of one question.",
'design-html-slop-gate':
"How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.",
'scrape-match-path':
"The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.",
'scrape-prototype-path':
"The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.",
};