mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 18:06:54 +02:00
test(evals): classify every live case and re-select a case when its kind changes
E2E_KINDS: rule by default (191 E2E ids), 22 behavior cases whose verdict is a live model choice with an acceptable sub-100% per-trial rate, each with a BEHAVIOR_WHY tolerance, and 25 judge entries (the 24 workflow judges plus the fixed-fixture llm-judge-recommendation rubric check). Contract-shaped cases (ask-before-decide, plan-mode no-writes, mandated steps, secrets, the batching floor) stay rule. Behavior requires a known literal registration and an exact Bun test name so the case runs as its own trial shard. Map-diff selection now diffs E2E_KINDS and BEHAVIOR_WHY per key, and a base revision without them selects every key, so a kind flip runs the panel it introduces. test/eval-kinds.test.ts enforces coverage, tolerances, isolatability and the reviewed counts, printing the literal to add.
This commit is contained in:
1 parent
3f68572cab
commit
0292ee3f6f
3 files changed
+216
-51
No files matched your search
@@ -0,0 +1,107 @@
|
||||
/**
|
||||
* Eval kind registry (E2E_KINDS / BEHAVIOR_WHY in touchfiles-data.ts). The
|
||||
* kind fixes a case's trial policy before the run, so the registry must cover
|
||||
* every live case exactly once, every behavior case must name its tolerated
|
||||
* deviation, and a behavior case must be isolatable as its own trial shard.
|
||||
* A kind edit must re-select the case in the PR lane (map-diff).
|
||||
*/
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
|
||||
import { BEHAVIOR_WHY, E2E_KINDS, E2E_TIERS, E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES } from './helpers/touchfiles-data';
|
||||
import { diffTouchfileMapsCore, type TouchfileMaps } from './helpers/test-selection';
|
||||
import { CASE_TEST_NAMES, fileCaseRegistration } from '../scripts/test-paid-shards';
|
||||
import { isPaidTestFile } from './helpers/paid-test-set';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const KIND_RULE = "Pick the kind by what can make the verdict differ between two runs of the same commit: 'rule' when nothing "
|
||||
+ "stochastic decides it or it checks a contract the product must meet every run (the default); 'behavior' when a live "
|
||||
+ "model choice decides it and a sub-100% per-trial rate is acceptable (add a BEHAVIOR_WHY line); 'judge' when the only "
|
||||
+ 'stochastic step is an LLM judge scoring a fixed input.';
|
||||
|
||||
const liveIds = [...Object.keys(E2E_TIERS), ...Object.keys(LLM_JUDGE_TOUCHFILES)];
|
||||
const behaviorIds = Object.keys(E2E_KINDS).filter(id => E2E_KINDS[id] === 'behavior').sort();
|
||||
|
||||
describe('E2E_KINDS registry', () => {
|
||||
test('every live case has exactly one kind and no kind names a dead case', () => {
|
||||
const missing = liveIds.filter(id => !(id in E2E_KINDS));
|
||||
expect(missing.length, missing.length ? `add to E2E_KINDS:\n${missing.map(id => ` '${id}': 'rule', // <reason>`).join('\n')}\n${KIND_RULE}` : '').toBe(0);
|
||||
const unknown = Object.keys(E2E_KINDS).filter(id => !liveIds.includes(id));
|
||||
expect(unknown, `E2E_KINDS names ids that are neither E2E_TIERS nor LLM_JUDGE_TOUCHFILES keys`).toEqual([]);
|
||||
expect(new Set(liveIds).size).toBe(liveIds.length);
|
||||
});
|
||||
|
||||
test('kinds are rule, behavior or judge; every LLM-judge entry is judge-kind', () => {
|
||||
for (const [id, kind] of Object.entries(E2E_KINDS)) expect(['rule', 'behavior', 'judge'], id).toContain(kind);
|
||||
for (const id of Object.keys(LLM_JUDGE_TOUCHFILES)) expect(E2E_KINDS[id], `${id}: a workflow judge scores a fixed input`).toBe('judge');
|
||||
});
|
||||
|
||||
test('BEHAVIOR_WHY names the tolerance of exactly the behavior cases', () => {
|
||||
expect(Object.keys(BEHAVIOR_WHY).sort()).toEqual(behaviorIds);
|
||||
for (const id of behaviorIds) {
|
||||
expect(BEHAVIOR_WHY[id]!.trim().length, `${id}: BEHAVIOR_WHY must say why an occasional deviation is acceptable`).toBeGreaterThanOrEqual(30);
|
||||
}
|
||||
});
|
||||
|
||||
test('a behavior case is an isolatable trial shard: known literal registration and an exact Bun test name', () => {
|
||||
for (const id of behaviorIds) {
|
||||
const files = E2E_TOUCHFILES[id]!.filter(file => /^test\/[^/]+\.test\.ts$/.test(file) && isPaidTestFile(file));
|
||||
expect(files.length, `${id}: no paid test file registers it`).toBeGreaterThan(0);
|
||||
for (const file of files) {
|
||||
const source = fs.readFileSync(path.join(ROOT, file), 'utf8');
|
||||
expect(fileCaseRegistration(file, source).known, `${id}: ${file} has a computed registration; behavior needs a literal one`).toBe(true);
|
||||
const name = CASE_TEST_NAMES[id] ?? id;
|
||||
const literal = new RegExp(`\\b(?:test(?:\\.serial|\\.concurrent)?|testIfSelected|testConcurrentIfSelected)\\(\\s*(['"\`])${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\1`);
|
||||
expect(literal.test(source), `${id}: ${file} must register the Bun test named '${name}'`).toBe(true);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test('the classification is the reviewed one: rule by default, 22 behavior, 25 judge', () => {
|
||||
const counts = Object.values(E2E_KINDS).reduce<Record<string, number>>((acc, kind) => ({ ...acc, [kind]: (acc[kind] ?? 0) + 1 }), {});
|
||||
expect(counts).toEqual({ rule: liveIds.length - 22 - 25, behavior: 22, judge: 25 });
|
||||
// Contract-shaped cases stay rule: ask-before-decide, plan-mode no-writes,
|
||||
// mandated steps, secrets, and the batching floor never ride a majority.
|
||||
for (const id of ['plan-ceo-mode-routing', 'plan-eng-multi-finding-batching', 'plan-design-review-plan-mode',
|
||||
'plan-eng-review-plan-mode', 'plan-ceo-section-loading', 'setup-gbrain-bad-token', 'qa-only-no-fix', 'review-sql-injection']) {
|
||||
expect(E2E_KINDS[id], id).toBe('rule');
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('kind edits re-select their case (map-diff)', () => {
|
||||
const base = (): TouchfileMaps => ({
|
||||
E2E_TOUCHFILES: { alpha: ['a/**'], beta: ['b/**'] },
|
||||
E2E_TIERS: { alpha: 'gate', beta: 'periodic' },
|
||||
LLM_JUDGE_TOUCHFILES: { 'judge one': ['j/SKILL.md'] },
|
||||
GLOBAL_TOUCHFILES: [],
|
||||
E2E_KINDS: { alpha: 'rule', beta: 'rule', 'judge one': 'judge' },
|
||||
BEHAVIOR_WHY: {},
|
||||
});
|
||||
|
||||
test('a rule -> behavior flip selects exactly that case', () => {
|
||||
const next = base();
|
||||
next.E2E_KINDS = { ...next.E2E_KINDS, beta: 'behavior' };
|
||||
next.BEHAVIOR_WHY = { beta: 'tolerated deviation' };
|
||||
expect(diffTouchfileMapsCore(base(), next).changedTests).toEqual(['beta']);
|
||||
});
|
||||
|
||||
test('a BEHAVIOR_WHY edit alone selects its case', () => {
|
||||
const old = base(); old.E2E_KINDS!.beta = 'behavior'; old.BEHAVIOR_WHY = { beta: 'one' };
|
||||
const next = base(); next.E2E_KINDS!.beta = 'behavior'; next.BEHAVIOR_WHY = { beta: 'two' };
|
||||
expect(diffTouchfileMapsCore(old, next).changedTests).toEqual(['beta']);
|
||||
});
|
||||
|
||||
test('a base revision without the kind maps selects every key', () => {
|
||||
const old = base(); delete old.E2E_KINDS; delete old.BEHAVIOR_WHY;
|
||||
expect(diffTouchfileMapsCore(old, base()).changedTests).toEqual(['alpha', 'beta', 'judge one']);
|
||||
});
|
||||
|
||||
test('dropping a kind entry while the case lives on counts as changed, not removed', () => {
|
||||
const next = base(); delete next.E2E_KINDS!.alpha;
|
||||
const result = diffTouchfileMapsCore(base(), next);
|
||||
expect(result.changedTests).toEqual(['alpha']);
|
||||
expect(result.removedTests).toEqual([]);
|
||||
});
|
||||
});
|
||||
@@ -34,6 +34,8 @@ import {
|
||||
E2E_TIERS,
|
||||
LLM_JUDGE_TOUCHFILES,
|
||||
GLOBAL_TOUCHFILES,
|
||||
E2E_KINDS,
|
||||
BEHAVIOR_WHY,
|
||||
} from './touchfiles-data';
|
||||
|
||||
/** Repo-relative path of the pure-data file (the map-diff subject). */
|
||||
@@ -145,6 +147,9 @@ export interface TouchfileMaps {
|
||||
E2E_TIERS: Record<string, string>;
|
||||
LLM_JUDGE_TOUCHFILES: Record<string, string[]>;
|
||||
GLOBAL_TOUCHFILES: string[];
|
||||
/** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */
|
||||
E2E_KINDS?: Record<string, string>;
|
||||
BEHAVIOR_WHY?: Record<string, string>;
|
||||
}
|
||||
|
||||
export type MapDiffCause =
|
||||
@@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = {
|
||||
E2E_TIERS,
|
||||
LLM_JUDGE_TOUCHFILES,
|
||||
GLOBAL_TOUCHFILES,
|
||||
E2E_KINDS,
|
||||
BEHAVIOR_WHY,
|
||||
};
|
||||
|
||||
function isStringArray(v: unknown): v is string[] {
|
||||
@@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps {
|
||||
return isRecordOfStringArrays(o.E2E_TOUCHFILES)
|
||||
&& isRecordOfStrings(o.E2E_TIERS)
|
||||
&& isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES)
|
||||
&& isStringArray(o.GLOBAL_TOUCHFILES);
|
||||
&& isStringArray(o.GLOBAL_TOUCHFILES)
|
||||
&& (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS))
|
||||
&& (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY));
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure map-diff core (injectable for tests — no git, no filesystem).
|
||||
*
|
||||
* A key counts as CHANGED when it was added to any per-key map, its dep-list
|
||||
* array differs, or its tier value flipped. A key counts as REMOVED only when
|
||||
* array differs, or its tier, kind or behavior tolerance changed. A per-key
|
||||
* map missing on the old side (a base revision older than E2E_KINDS /
|
||||
* BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when
|
||||
* it is gone from every new per-key map; a key dropped from one map but still
|
||||
* present in another (e.g. tier entry deleted, touchfile entry kept) counts
|
||||
* as changed — conservative, because the test still exists with a different
|
||||
@@ -212,7 +223,7 @@ export function diffTouchfileMapsCore(
|
||||
oldMaps: TouchfileMaps,
|
||||
newMaps: TouchfileMaps,
|
||||
): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } {
|
||||
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const;
|
||||
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const;
|
||||
const changed = new Set<string>();
|
||||
const rawRemoved = new Set<string>();
|
||||
|
||||
@@ -290,6 +301,8 @@ export function diffTouchfileMaps(
|
||||
' E2E_TIERS: m.E2E_TIERS,',
|
||||
' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,',
|
||||
' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,',
|
||||
' E2E_KINDS: m.E2E_KINDS,',
|
||||
' BEHAVIOR_WHY: m.BEHAVIOR_WHY,',
|
||||
'}));',
|
||||
'',
|
||||
].join('\n'));
|
||||
|
||||
@@ -1597,7 +1597,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'shared-libs-unsupported-git': 'rule',
|
||||
'shared-libs-review-lifecycle': 'rule',
|
||||
'shared-libs-review-revalidation': 'rule',
|
||||
'shared-libs-opportunity-judgment': 'rule',
|
||||
'shared-libs-opportunity-judgment': 'behavior',
|
||||
'shared-libs-pr-coverage': 'rule',
|
||||
'shared-libs-plan-callers': 'rule',
|
||||
'browse-basic': 'rule',
|
||||
@@ -1634,7 +1634,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'review-sql-injection': 'rule',
|
||||
'review-enum-completeness': 'rule',
|
||||
'review-base-branch': 'rule',
|
||||
'review-design-lite': 'rule',
|
||||
'review-design-lite': 'behavior',
|
||||
'review-coverage-audit': 'rule',
|
||||
'review-dashboard-via': 'rule',
|
||||
'review-army-migration-safety': 'rule',
|
||||
@@ -1642,21 +1642,21 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'review-army-delivery-audit': 'rule',
|
||||
'review-army-quality-score': 'rule',
|
||||
'review-army-json-findings': 'rule',
|
||||
'review-army-red-team': 'rule',
|
||||
'review-army-consensus': 'rule',
|
||||
'review-army-simplification': 'rule',
|
||||
'review-army-simplification-precision': 'rule',
|
||||
'review-army-red-team': 'behavior',
|
||||
'review-army-consensus': 'behavior',
|
||||
'review-army-simplification': 'behavior',
|
||||
'review-army-simplification-precision': 'behavior',
|
||||
'office-hours-spec-review': 'rule',
|
||||
'office-hours-brain-writeback': 'rule',
|
||||
'office-hours-brain-writeback': 'behavior',
|
||||
'gbrain-roundtrip-local': 'rule',
|
||||
'sync-gbrain-read-ready': 'rule',
|
||||
'sync-gbrain-read-unknown': 'rule',
|
||||
'office-hours-forcing-energy': 'rule',
|
||||
'office-hours-builder-wildness': 'rule',
|
||||
'office-hours-forcing-energy': 'behavior',
|
||||
'office-hours-builder-wildness': 'behavior',
|
||||
'plan-ceo-review': 'rule',
|
||||
'plan-ceo-review-selective': 'rule',
|
||||
'plan-ceo-review-benefits': 'rule',
|
||||
'plan-ceo-review-expansion-energy': 'rule',
|
||||
'plan-ceo-review-expansion-energy': 'behavior',
|
||||
'plan-eng-review': 'rule',
|
||||
'plan-eng-review-artifact': 'rule',
|
||||
'plan-eng-coverage-audit': 'rule',
|
||||
@@ -1688,16 +1688,16 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'setup-gbrain-remote': 'rule',
|
||||
'setup-gbrain-bad-token': 'rule',
|
||||
'setup-gbrain-path4-local-pglite': 'rule',
|
||||
'plan-ceo-review-format-mode': 'rule',
|
||||
'plan-ceo-review-format-approach': 'rule',
|
||||
'plan-eng-review-format-coverage': 'rule',
|
||||
'plan-eng-review-format-kind': 'rule',
|
||||
'office-hours-phase4-fork': 'rule',
|
||||
'llm-judge-recommendation': 'rule',
|
||||
'plan-ceo-review-prosons-cadence': 'rule',
|
||||
'plan-review-prosons-format': 'rule',
|
||||
'plan-review-prosons-hardstop-neg': 'rule',
|
||||
'plan-review-prosons-neutral-neg': 'rule',
|
||||
'plan-ceo-review-format-mode': 'behavior',
|
||||
'plan-ceo-review-format-approach': 'behavior',
|
||||
'plan-eng-review-format-coverage': 'behavior',
|
||||
'plan-eng-review-format-kind': 'behavior',
|
||||
'office-hours-phase4-fork': 'behavior',
|
||||
'llm-judge-recommendation': 'judge',
|
||||
'plan-ceo-review-prosons-cadence': 'behavior',
|
||||
'plan-review-prosons-format': 'behavior',
|
||||
'plan-review-prosons-hardstop-neg': 'behavior',
|
||||
'plan-review-prosons-neutral-neg': 'behavior',
|
||||
'plan-tune-inspect': 'rule',
|
||||
'codex-offered-office-hours': 'rule',
|
||||
'codex-offered-ceo-review': 'rule',
|
||||
@@ -1758,7 +1758,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'design-review-detector-shim': 'rule',
|
||||
'design-review-detector-shim-dom': 'rule',
|
||||
'design-review-plugin-handoff': 'rule',
|
||||
'design-html-slop-gate': 'rule',
|
||||
'design-html-slop-gate': 'behavior',
|
||||
'diagram-triplet': 'rule',
|
||||
'diagram-authoring-quality': 'rule',
|
||||
'gstack-upgrade-happy-path': 'rule',
|
||||
@@ -1770,8 +1770,8 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'setup-deploy-workflow': 'rule',
|
||||
'autoplan-dual-voice': 'rule',
|
||||
'benchmark-providers-live': 'rule',
|
||||
'scrape-match-path': 'rule',
|
||||
'scrape-prototype-path': 'rule',
|
||||
'scrape-match-path': 'behavior',
|
||||
'scrape-prototype-path': 'behavior',
|
||||
'skillify-happy-path': 'rule',
|
||||
'skillify-provenance-refusal': 'rule',
|
||||
'skillify-approval-reject': 'rule',
|
||||
@@ -1799,30 +1799,30 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'overlay-harness-opus-4-7-literal-interpretation': 'rule',
|
||||
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule',
|
||||
'journey-negatives': 'rule',
|
||||
'review/SKILL.md workflow': 'rule',
|
||||
'setup-browser-cookies/SKILL.md workflow': 'rule',
|
||||
'browse/SKILL.md reference': 'rule',
|
||||
'setup block': 'rule',
|
||||
'qa/SKILL.md workflow': 'rule',
|
||||
'qa/SKILL.md health rubric': 'rule',
|
||||
'qa/SKILL.md anti-refusal': 'rule',
|
||||
'cross-skill greptile consistency': 'rule',
|
||||
'ship/SKILL.md workflow': 'rule',
|
||||
'document-release/SKILL.md workflow': 'rule',
|
||||
'plan-ceo-review/SKILL.md modes': 'rule',
|
||||
'plan-eng-review/SKILL.md sections': 'rule',
|
||||
'plan-design-review/SKILL.md passes': 'rule',
|
||||
'design-review/SKILL.md fix loop': 'rule',
|
||||
'design-consultation/SKILL.md research': 'rule',
|
||||
'land-and-deploy/SKILL.md workflow': 'rule',
|
||||
'canary/SKILL.md monitoring loop': 'rule',
|
||||
'benchmark/SKILL.md perf collection': 'rule',
|
||||
'setup-deploy/SKILL.md platform setup': 'rule',
|
||||
'retro/SKILL.md instructions': 'rule',
|
||||
'qa-only/SKILL.md workflow': 'rule',
|
||||
'gstack-upgrade/SKILL.md upgrade flow': 'rule',
|
||||
'sync-gbrain/SKILL.md read-only readiness': 'rule',
|
||||
'voice directive tone': 'rule',
|
||||
'review/SKILL.md workflow': 'judge',
|
||||
'setup-browser-cookies/SKILL.md workflow': 'judge',
|
||||
'browse/SKILL.md reference': 'judge',
|
||||
'setup block': 'judge',
|
||||
'qa/SKILL.md workflow': 'judge',
|
||||
'qa/SKILL.md health rubric': 'judge',
|
||||
'qa/SKILL.md anti-refusal': 'judge',
|
||||
'cross-skill greptile consistency': 'judge',
|
||||
'ship/SKILL.md workflow': 'judge',
|
||||
'document-release/SKILL.md workflow': 'judge',
|
||||
'plan-ceo-review/SKILL.md modes': 'judge',
|
||||
'plan-eng-review/SKILL.md sections': 'judge',
|
||||
'plan-design-review/SKILL.md passes': 'judge',
|
||||
'design-review/SKILL.md fix loop': 'judge',
|
||||
'design-consultation/SKILL.md research': 'judge',
|
||||
'land-and-deploy/SKILL.md workflow': 'judge',
|
||||
'canary/SKILL.md monitoring loop': 'judge',
|
||||
'benchmark/SKILL.md perf collection': 'judge',
|
||||
'setup-deploy/SKILL.md platform setup': 'judge',
|
||||
'retro/SKILL.md instructions': 'judge',
|
||||
'qa-only/SKILL.md workflow': 'judge',
|
||||
'gstack-upgrade/SKILL.md upgrade flow': 'judge',
|
||||
'sync-gbrain/SKILL.md read-only readiness': 'judge',
|
||||
'voice directive tone': 'judge',
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -1830,4 +1830,49 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
* deviation is acceptable product behavior. Keys equal the behavior ids of
|
||||
* E2E_KINDS; values are non-empty.
|
||||
*/
|
||||
export const BEHAVIOR_WHY: Record<string, string> = {};
|
||||
export const BEHAVIOR_WHY: Record<string, string> = {
|
||||
'shared-libs-opportunity-judgment':
|
||||
"Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.",
|
||||
'review-design-lite':
|
||||
"How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.",
|
||||
'review-army-red-team':
|
||||
"Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.",
|
||||
'review-army-consensus':
|
||||
"Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.",
|
||||
'review-army-simplification':
|
||||
"Flagging the planted unnecessary structure is an advisory-lens quality judgment.",
|
||||
'review-army-simplification-precision':
|
||||
"Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.",
|
||||
'office-hours-forcing-energy':
|
||||
"The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.",
|
||||
'office-hours-builder-wildness':
|
||||
"Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.",
|
||||
'office-hours-brain-writeback':
|
||||
"The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.",
|
||||
'office-hours-phase4-fork':
|
||||
"Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.",
|
||||
'plan-ceo-review-expansion-energy':
|
||||
"Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.",
|
||||
'plan-ceo-review-format-mode':
|
||||
"Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.",
|
||||
'plan-ceo-review-format-approach':
|
||||
"Approach-menu Completeness wording is live formatting of one AskUserQuestion.",
|
||||
'plan-eng-review-format-coverage':
|
||||
"Coverage-issue Completeness wording is live formatting of one AskUserQuestion.",
|
||||
'plan-eng-review-format-kind':
|
||||
"Kind-note wording is live formatting of one AskUserQuestion.",
|
||||
'plan-ceo-review-prosons-cadence':
|
||||
"Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.",
|
||||
'plan-review-prosons-format':
|
||||
"The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.",
|
||||
'plan-review-prosons-hardstop-neg':
|
||||
"Not using the hard-stop escape on an ordinary decision is live formatting of one question.",
|
||||
'plan-review-prosons-neutral-neg':
|
||||
"Avoiding neutral posture and naming a because-reason is live formatting of one question.",
|
||||
'design-html-slop-gate':
|
||||
"How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.",
|
||||
'scrape-match-path':
|
||||
"The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.",
|
||||
'scrape-prototype-path':
|
||||
"The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.",
|
||||
};
|
||||
Reference in new issue
Block a user