test: clean up the paid eval lane (B1-B4, B6, B7)

- B1: delete paid files that assert nothing or cannot pass meaningfully:
  skill-llm-eval-spec and skill-e2e-spec-execute (test.todo), gemini-e2e
  (+ gemini-session-runner; no gemini CLI in CI), ship-idempotency (red
  since v1.63), the two opus-4-7 *-sonnet overlay wrappers, conductor-prose
  (+ its source-evaluation replay), codex-e2e-plan-format; drop their keys,
  scripts and census rows.
- B2: skill-llm-eval grades browse/sections/command-list.md with one union
  judge that also carries the baseline score pin; regression-vs-baseline
  deleted (paid run: pass, c4/c4/a4).
- B3: memory-pipeline, ios-qa, ios-qa-swift-build and plan-tune-cathedral
  make no model calls; renamed out of the paid glob so they run on every
  PR. Swift builds need GSTACK_TEST_SWIFT=1; device stub deleted.
- B4: codex-e2e*, outside-voice, aside and ios-device cannot run in the CI
  image; excluded from the weekly lane with a tracked re-entry condition.
- B6: fold opus-47's negative routing controls into skill-routing-e2e
  journey-negatives (paid run: 3/3 unrouted) and delete the file.
- B7: delete the never-green brain-privacy-gate eval; a free
  gstack-skill-start test now proves consent precedes artifacts egress.
This commit is contained in:
garrytan committed 2026-09-29 06:08:49 +00:00
1 parent 5d032ef299
commit 53e7f3212f
58 files changed
+273 -2594

No files matched your search

+1 -1
View File
@@ -50,7 +50,7 @@ test('failed loads, foreign sessions, actual questions and later withdrawals ret
import { E2E_TOUCHFILES } from './helpers/touchfiles-data';
test('new evidence inputs retain every native observation caller',()=>{
const owners=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','auto-decide-preserved','conductor-prose'];
const owners=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','auto-decide-preserved'];
for(const file of ['test/auto-decide-saved-ai.test.ts','test/fixtures/auto-decide-saved-ai.json','test/fixtures/auto-decide-retry-ai.json'])
expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(file)).map(([name])=>name)).toEqual(owners);
});
+1 -1
View File
@@ -127,7 +127,7 @@ test('phase ordering and duplicate collapse use native time rather than polling
test('public narration changes select every existing shared native-reader consumer',()=>{
const expected=[
'auto-decide-preserved','autoplan-chain-pty','conductor-prose',
'auto-decide-preserved','autoplan-chain-pty',
'plan-ceo-finding-count','plan-ceo-mode-routing','plan-ceo-split-overflow',
'plan-design-finding-count','plan-design-review-plan-mode','plan-design-with-ui-scope',
'plan-devex-finding-count','plan-eng-finding-count','plan-eng-multi-finding-batching',
-289
View File
@@ -1,289 +0,0 @@
/**
* AskUserQuestion format regression test for /plan-ceo-review and /plan-eng-review
* running under Codex CLI (GPT-5.4).
*
* Context: GPT-class models under the "No preamble / Prefer doing over listing"
* gpt.md overlay tend to skip the Simplify (ELI10) paragraph and the RECOMMENDATION
* line on AskUserQuestion calls. The user has to manually re-prompt "ELI10 and don't
* forget to recommend" almost every time. This test pins that behavior so future
* regressions surface automatically.
*
* Mirrors test/skill-e2e-plan-format.test.ts (the Claude version) but uses
* test/helpers/codex-session-runner.ts to drive `codex exec` instead of `claude -p`.
*
* Four cases:
* 1. plan-ceo-review mode selection (kind-differentiated)
* 2. plan-ceo-review approach menu (coverage-differentiated)
* 3. plan-eng-review per-issue coverage decision
* 4. plan-eng-review per-issue architectural choice (kind-differentiated)
*
* Assertions on captured AskUserQuestion text:
* - RECOMMENDATION: Choose present (all cases)
* - Completeness: N/10 present on coverage cases, absent on kind cases
* - "options differ in kind" note present on kind cases
* - ELI10-style plain-English explanation present (length floor + no raw jargon)
*
* Periodic tier (Codex non-determinism). Cost: ~$2-3 per full run.
*/
import { describe, test, beforeAll, afterAll } from 'bun:test';
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
import { runCodexSkill } from './helpers/codex-session-runner';
import { CODEX_EVAL_FINALIZE_MS, createCodexEvalCollector, runRecordedCodexEval, createCodexPlanFormatCapture } from './helpers/codex-eval';
import { selectTests, detectBaseBranch, getChangedFiles, E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from './helpers/touchfiles';
import * as fs from 'fs';
import * as path from 'path';
import * as os from 'os';
import { spawnSync } from 'child_process';
const ROOT = path.resolve(import.meta.dir, '..');
// --- Prerequisites ---
const CODEX_AVAILABLE = (() => {
try {
const result = Bun.spawnSync(['which', 'codex'], { timeout: 30_000 });
return result.exitCode === 0;
} catch { return false; }
})();
const evalsEnabled = !!process.env.EVALS;
// External-service test — periodic tier only (CLAUDE.md tiering rule 3),
// matching codex-e2e.test.ts / codex-e2e-sol-scope.test.ts. Without this
// guard the sharded runner's "no whole-file tier guard" default would run
// Codex spawns in the GATE tier on every PR.
const tierOk = process.env.EVALS_TIER === 'periodic';
const SKIP = !CODEX_AVAILABLE || !evalsEnabled || !tierOk;
const describeCodex = SKIP ? describe.skip : describe;
// --- Touchfiles ---
// Keep selection dependencies in the canonical map, including the test helpers.
const CODEX_FORMAT_TOUCHFILES: Record<string, string[]> = Object.fromEntries(
['codex-plan-ceo-format-mode', 'codex-plan-ceo-format-approach',
'codex-plan-eng-format-coverage', 'codex-plan-eng-format-kind'].map((key) => {
if (!E2E_TOUCHFILES[key]) throw new Error(`canonical E2E_TOUCHFILES lost key '${key}'`);
return [key, E2E_TOUCHFILES[key]];
}),
);
let selectedTests: string[] | null = null;
if (evalsEnabled && !process.env.EVALS_ALL) {
const baseBranch = process.env.EVALS_BASE || detectBaseBranch(ROOT) || 'main';
const changedFiles = getChangedFiles(baseBranch, ROOT);
if (changedFiles.length > 0) {
const selection = selectTests(changedFiles, CODEX_FORMAT_TOUCHFILES, GLOBAL_TOUCHFILES);
selectedTests = selection.selected;
}
}
function testIfSelected(name: string, fn: () => Promise<void>, timeout: number) {
if (selectedTests !== null && !selectedTests.includes(name)) {
test.skip(name, fn, timeout + CODEX_EVAL_FINALIZE_MS);
} else {
test(name, fn, timeout + CODEX_EVAL_FINALIZE_MS);
}
}
// --- Eval collector ---
const evalCollector = SKIP ? null : createCodexEvalCollector('codex-e2e-plan-format');
afterAll(async () => {
if (evalCollector) {
await evalCollector.finalize();
}
});
// --- Fixtures ---
const SAMPLE_PLAN = `# Plan: Add User Dashboard
## Context
We're building a new user dashboard that shows recent activity, notifications, and quick actions.
## Changes
1. New React component \`UserDashboard\` in \`src/components/\`
2. REST API endpoint \`GET /api/dashboard\` returning user stats
3. PostgreSQL query for activity aggregation
4. Redis cache layer for dashboard data (5min TTL)
## Architecture
- Frontend: React + TailwindCSS
- Backend: Express.js REST API
- Database: PostgreSQL with existing user/activity tables
- Cache: Redis for dashboard aggregates
`;
function setupCodexSkillDir(tmpPrefix: string, skillName: 'plan-ceo-review' | 'plan-eng-review'): { skillDir: string; planDir: string; outFile: string } {
const planDir = fs.mkdtempSync(path.join(os.tmpdir(), tmpPrefix));
const run = (cmd: string, args: string[]) =>
spawnSync(cmd, args, { cwd: planDir, stdio: 'pipe', timeout: 5000 });
run('git', ['init', '-b', 'main']);
run('git', ['config', 'user.email', 'test@test.com']);
run('git', ['config', 'user.name', 'Test']);
fs.writeFileSync(path.join(planDir, 'plan.md'), SAMPLE_PLAN);
run('git', ['add', '.']);
run('git', ['commit', '-m', 'add plan']);
// Codex skill lives in .agents/skills/gstack-{name}/ per the gstack host convention.
const codexSkillSource = path.join(ROOT, '.agents', 'skills', `gstack-${skillName}`);
const skillDir = path.join(planDir, '.agents', 'skills', `gstack-${skillName}`);
fs.mkdirSync(skillDir, { recursive: true });
fs.cpSync(codexSkillSource, skillDir, { recursive: true });
const outFile = path.join(planDir, 'ask-capture.md');
return { skillDir, planDir, outFile };
}
// Capture instruction — same shape as the Claude version. Codex may ignore tool calls,
// so we tell it to write prose to the file directly.
function captureInstruction(outFile: string): string {
return `Write the verbatim text of every AskUserQuestion you would have presented to the user to the file ${outFile} (one question per session, full text including the re-ground, ELI10 paragraph, RECOMMENDATION line, and options). Do NOT ask the user interactively. Do NOT paraphrase. This is a format-capture test, not an interactive session.`;
}
// --- Tests ---
describeCodex('Codex Plan Format — CEO Mode Selection', () => {
let skillDir: string, planDir: string, outFile: string;
beforeAll(() => {
({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-ceo-mode-', 'plan-ceo-review'));
});
afterAll(() => {
try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {}
});
testIfSelected('codex-plan-ceo-format-mode', async () => {
const capture = createCodexPlanFormatCapture(outFile, 'kind');
const result = await runRecordedCodexEval({
name: 'codex-plan-ceo-format-mode',
suite: 'codex-e2e-plan-format',
budgetMs: CAPTURE_LONG_MS,
run: (signal) => {
capture.reset();
return runCodexSkill({
skillDir,
prompt: `Read the plan-ceo-review skill. Read plan.md (the plan to review). Proceed to Mode Selection where the skill presents 4 mode options (SCOPE EXPANSION, SELECTIVE EXPANSION, HOLD SCOPE, SCOPE REDUCTION) via AskUserQuestion. These options differ in kind (review posture), not coverage. ${captureInstruction(outFile)}`,
timeoutMs: CAPTURE_MS,
cwd: planDir,
skillName: 'gstack-plan-ceo-review',
sandbox: 'workspace-write',
signal,
});
},
validate: capture.validate,
record: (entry) => evalCollector?.addTest(capture.attach(entry)),
});
console.log(`codex-plan-ceo-format-mode: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`);
}, CAPTURE_LONG_MS);
});
describeCodex('Codex Plan Format — CEO Approach Menu', () => {
let skillDir: string, planDir: string, outFile: string;
beforeAll(() => {
({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-ceo-approach-', 'plan-ceo-review'));
});
afterAll(() => {
try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {}
});
testIfSelected('codex-plan-ceo-format-approach', async () => {
const capture = createCodexPlanFormatCapture(outFile, 'coverage');
const result = await runRecordedCodexEval({
name: 'codex-plan-ceo-format-approach',
suite: 'codex-e2e-plan-format',
budgetMs: CAPTURE_LONG_MS,
run: (signal) => {
capture.reset();
return runCodexSkill({
skillDir,
prompt: `Read the plan-ceo-review skill. Read plan.md. Proceed to Alternatives (the implementation approach menu) where the skill generates 2-3 approaches (minimal viable vs ideal architecture) and presents them via AskUserQuestion. These options differ in coverage so Completeness: N/10 applies. ${captureInstruction(outFile)}`,
timeoutMs: CAPTURE_MS,
cwd: planDir,
skillName: 'gstack-plan-ceo-review',
sandbox: 'workspace-write',
signal,
});
},
validate: capture.validate,
record: (entry) => evalCollector?.addTest(capture.attach(entry)),
});
console.log(`codex-plan-ceo-format-approach: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`);
}, CAPTURE_LONG_MS);
});
describeCodex('Codex Plan Format — Eng Coverage Issue', () => {
let skillDir: string, planDir: string, outFile: string;
beforeAll(() => {
({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-eng-cov-', 'plan-eng-review'));
});
afterAll(() => {
try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {}
});
testIfSelected('codex-plan-eng-format-coverage', async () => {
const capture = createCodexPlanFormatCapture(outFile, 'coverage');
const result = await runRecordedCodexEval({
name: 'codex-plan-eng-format-coverage',
suite: 'codex-e2e-plan-format',
budgetMs: CAPTURE_LONG_MS,
run: (signal) => {
capture.reset();
return runCodexSkill({
skillDir,
prompt: `Read the plan-eng-review skill. Read plan.md. In your Section 3 Test Review, generate ONE AskUserQuestion about test coverage depth where options are clearly coverage-differentiated: A) full coverage incl. edge + error paths (Completeness 10/10), B) happy path only (7/10), C) smoke test (3/10). ${captureInstruction(outFile)}`,
timeoutMs: CAPTURE_MS,
cwd: planDir,
skillName: 'gstack-plan-eng-review',
sandbox: 'workspace-write',
signal,
});
},
validate: capture.validate,
record: (entry) => evalCollector?.addTest(capture.attach(entry)),
});
console.log(`codex-plan-eng-format-coverage: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`);
}, CAPTURE_LONG_MS);
});
describeCodex('Codex Plan Format — Eng Kind Issue', () => {
let skillDir: string, planDir: string, outFile: string;
beforeAll(() => {
({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-eng-kind-', 'plan-eng-review'));
});
afterAll(() => {
try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {}
});
testIfSelected('codex-plan-eng-format-kind', async () => {
const capture = createCodexPlanFormatCapture(outFile, 'kind');
const result = await runRecordedCodexEval({
name: 'codex-plan-eng-format-kind',
suite: 'codex-e2e-plan-format',
budgetMs: CAPTURE_LONG_MS,
run: (signal) => {
capture.reset();
return runCodexSkill({
skillDir,
prompt: `Read the plan-eng-review skill. Read plan.md. In your Section 1 Architecture review, generate ONE AskUserQuestion about an architectural choice where the options differ in kind (e.g. Redis vs Postgres materialized view vs in-process cache — different kinds of systems with different tradeoffs, NOT more-or-less-complete versions of the same thing). ${captureInstruction(outFile)}`,
timeoutMs: CAPTURE_MS,
cwd: planDir,
skillName: 'gstack-plan-eng-review',
sandbox: 'workspace-write',
signal,
});
},
validate: capture.validate,
record: (entry) => evalCollector?.addTest(capture.attach(entry)),
});
console.log(`codex-plan-eng-format-kind: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`);
}, CAPTURE_LONG_MS);
});
-12
View File
@@ -11,23 +11,11 @@ describe('Codex eval selection', () => {
expect(selected.sort()).toEqual([
'codex-discover-skill',
'codex-review-findings',
'codex-plan-ceo-format-mode',
'codex-plan-ceo-format-approach',
'codex-plan-eng-format-coverage',
'codex-plan-eng-format-kind',
'codex-sol-scope-termination',
].sort());
expect(selected.every((id) => E2E_TIERS[id] === 'periodic')).toBe(true);
});
test('format cases are selected by canonical source and their own test file', () => {
expect(selectedBy('plan-ceo-review/SKILL.md.tmpl')).toContain('codex-plan-ceo-format-mode');
expect(selectedBy('plan-ceo-review/SKILL.md.tmpl')).toContain('codex-plan-ceo-format-approach');
expect(selectedBy('plan-eng-review/SKILL.md.tmpl')).toContain('codex-plan-eng-format-coverage');
expect(selectedBy('plan-eng-review/SKILL.md.tmpl')).toContain('codex-plan-eng-format-kind');
expect(selectedBy('test/codex-e2e-plan-format.test.ts').length).toBe(4);
});
test('Sol fixture generation changes select its periodic case', () => {
for (const file of ['test/helpers/sol-skill-fixture.ts', 'test/sol-skill-fixture.test.ts']) {
expect(selectedBy(file)).toEqual(['codex-sol-scope-termination']);
-103
View File
@@ -1,103 +0,0 @@
import { expect, test } from 'bun:test';
import fs from 'node:fs';
import path from 'node:path';
import * as predicates from './helpers/claude-pty-runner';
import { E2E_TOUCHFILES } from './helpers/touchfiles-data';
import fixture from './fixtures/conductor-prose-ao.json';
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
const partial=fixture.publicDecisionTail;
// This next frame is synthetic; the retained live attempt ended during A.
const complete=partial+'\nB) Keep all four components and define cache invalidation before implementation.\nReply with A or B.';
async function observe(frames:string[],verdict:'waiting'|'working',required?:boolean){
const source=fs.readFileSync(path.join(import.meta.dir,'helpers/claude-pty-runner.ts'),'utf8');
const start=source.indexOf('export async function runPlanSkillObservation('),end=source.indexOf('\n// ─',start);
expect(start).toBeGreaterThan(0);expect(end).toBeGreaterThan(start);
const js=new Bun.Transpiler({loader:'ts'}).transformSync(source.slice(start,end).replace('export async function','async function')+'\nreturn runPlanSkillObservation;');
let clock=0,tick=-1,closed=0,judged=0;
const current=()=>frames[Math.min(Math.max(tick,0),frames.length-1)]!;
const args:Record<string,unknown>={path,process:{cwd:()=>'/synthetic-owned'},Date:{now:()=>clock},randomUUID:()=> 'owned',
Bun:{sleep:async(ms:number)=>{if(ms===2000){tick++;clock+=61000;}else clock+=ms;}},
launchClaudePty:async()=>({send:()=>{},mark:()=>0,exited:()=>false,visibleSince:current,rawOutput:current,currentScreen:async()=>current(),hermeticConfigDir:null,close:async()=>{closed++;}}),
createPlanCountSnapshotWriter:()=>()=>({}),logPtySnapshot:()=>{},
submitPlanSeed: async () => {}, PlanSeedTimeout: class extends Error {},
isRejectedSlashCommand:predicates.isRejectedSlashCommand,
isProseAUQVisible:predicates.isProseAUQVisible,isPlanReadyVisible:predicates.isPlanReadyVisible,
isUnknownSlashCommandVisible:predicates.isUnknownSlashCommandVisible,
isScopeGateQuestionVisible:predicates.isScopeGateQuestionVisible,isScopeGateAutoSelectVisible:predicates.isScopeGateAutoSelectVisible,
classifyVisible:predicates.classifyVisible,extractPlanFilePath:predicates.extractPlanFilePath,findNativeAutoDecision:()=>null,
judgePtyState:()=>{judged++;return {state:verdict,reasoning:'synthetic fixed verdict'};},
};
const run=new Function(...Object.keys(args),js)(...Object.values(args));
const obs=await run({skillName:'plan-eng-review',initialPlanContent:'# Plan: Required draft',timeoutMs:300000,...(required===undefined?{}:{requireProseEvidence:required})});
expect(closed).toBe(1);
return {obs,polls:tick+1,judged};
}
test('a judge waiting on the exact partial Conductor brief cannot stop a prose-required observation',async()=>{
expect(fixture.actualFlags.proseAUQEverObserved).toBe(false);expect(fixture.actualFlags.waitingEverObserved).toBe(true);
expect(predicates.isProseAUQVisible(partial)).toBe(false);expect(predicates.isProseAUQVisible(complete)).toBe(true);
const {obs,polls,judged}=await observe([partial,complete],'waiting',true);
expect(polls).toBe(2);expect(judged).toBe(1);expect(obs.outcome).toBe('asked');
expect(obs.proseAUQEverObserved).toBe(true);expect(obs.waitingEverObserved).toBe(true);
});
test('partial-only judge waiting reaches the existing budget without gaining prose fallback credit',async()=>{
const {obs,polls,judged}=await observe([partial],'waiting',true);
expect(polls).toBe(5);expect(judged).toBe(5);expect(obs.outcome).toBe('timeout');
expect(obs.proseAUQEverObserved).toBe(false);expect(obs.waitingEverObserved).toBe(true);
});
test('the prose requirement does not alter completed questions or deterministic failure precedence',async()=>{
const done=await observe([complete],'waiting',true);
expect(done.obs.outcome).toBe('asked');expect(done.obs.proseAUQEverObserved).toBe(true);expect(done.judged).toBe(0);
const wrote=await observe(['⏺ Write(/tmp/foreign-output.md)'],'waiting',true);
expect(wrote.obs.outcome).toBe('silent_write');expect(wrote.obs.proseAUQEverObserved).toBe(false);expect(wrote.judged).toBe(0);
});
test('other callers retain the original judge waiting behavior',async()=>{
for(const required of [undefined,false]){
const {obs,polls}=await observe([partial,complete],'waiting',required);
expect(polls).toBe(1);expect(obs.outcome).toBe('asked');expect(obs.proseAUQEverObserved).toBe(false);expect(obs.waitingEverObserved).toBe(true);
}
const working=await observe([partial],'working',true);
expect(working.obs.outcome).toBe('timeout');expect(working.obs.waitingEverObserved).toBe(false);
});
async function exerciseCaller(outcome: string, proseAUQEverObserved: boolean) {
const caller=fs.readFileSync(path.join(import.meta.dir,'skill-e2e-conductor-prose.test.ts'),'utf8');
const callbacks: Array<() => Promise<void>> = [];
let calls = 0;
const bindings = {
expect, CAPTURE_MS, CAPTURE_LONG_MS,
describeE2ETier: (tier: string) => {
expect(tier).toBe('periodic');
return (_title: string, register: () => void) => register();
},
test: (_title: string, callback: () => Promise<void>, timeout: number) => {
expect(timeout).toBe(CAPTURE_LONG_MS); callbacks.push(callback);
},
runPlanSkillObservation: async (opts: Record<string, unknown>) => {
calls++;
expect(opts).toMatchObject({skillName: 'plan-eng-review', inPlanMode: true,
requireProseEvidence: true, timeoutMs: CAPTURE_MS,
env: {CONDUCTOR_WORKSPACE_PATH: '/tmp/conductor-prose-e2e'},
extraArgs: ['--disallowedTools', 'AskUserQuestion']});
return {outcome, proseAUQEverObserved, summary: 'controlled caller', evidence: partial};
},
};
const body = caller.replace(/^import[\s\S]*?;\n/gm, '');
new Function(...Object.keys(bindings), new Bun.Transpiler({loader: 'ts'}).transformSync(body))(...Object.values(bindings));
expect(callbacks).toHaveLength(1);
try { await callbacks[0]!(); }
finally { expect(calls).toBe(1); }
}
test('the actual Conductor caller requests prose evidence and accepts a completed decision', async () => {
await exerciseCaller('asked', true);
});
test.each(['asked', 'auto_decided', 'plan_ready'])('the actual Conductor caller rejects %s without independent prose evidence', async outcome => {
await expect(exerciseCaller(outcome, false)).rejects.toThrow('Conductor prose decision not observed');
});
test.each(['silent_write', 'timeout', 'exited'])('the actual Conductor caller preserves the %s failure even with earlier prose evidence', async outcome => {
await expect(exerciseCaller(outcome, true)).rejects.toThrow(outcome === 'silent_write'
? 'skill wrote findings without surfacing a decision' : `outcome=${outcome}`);
});
test('the Conductor regression and public fixture select their existing owner',()=>{
for(const p of ['test/conductor-prose-observation-ao.test.ts','test/fixtures/conductor-prose-ao.json'])expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(p)).map(([owner])=>owner)).toEqual(['conductor-prose']);
});
+2 -2
View File
@@ -41,8 +41,8 @@ test('the existing quality and behavior phases retain their complete separate sh
const behaviorFiles = behavior.entries.filter(entry => entry.status === 'planned').map(entry => entry.file);
expect(quality.evalsAll).toBe(true);
expect(behavior.evalsAll).toBe(true);
expect(qualityFiles).toHaveLength(2);
expect(behaviorFiles).toHaveLength(56);
expect(qualityFiles).toHaveLength(1);
expect(behaviorFiles).toHaveLength(51);
expect(qualityFiles.every(file => file.startsWith('test/skill-llm-eval'))).toBe(true);
expect(behaviorFiles.every(file => !qualityFiles.includes(file))).toBe(true);
});
+1 -1
View File
@@ -29,7 +29,7 @@ describe('AO completed manual DX handoff preserves report freshness',()=>{
expect(E2E_TOUCHFILES[owner]).toContain('test/fixtures/dx-manual-handoff-ao.json');
}
const arrays=[...Object.values(E2E_TOUCHFILES),...Object.values(LLM_JUDGE_TOUCHFILES),GLOBAL_TOUCHFILES];
expect(arrays).toHaveLength(244);
expect(arrays).toHaveLength(221);
for(const values of arrays)for(let i=0;i<values.length;i++)expect(typeof values[i]).toBe('string');
});
test('exact owned report precedes navigation only, with the current Exit gate recognized',()=>{
-2
View File
@@ -57,8 +57,6 @@ const KNOWN_UNREGISTERED = new Set([
'test/skill-e2e-auq-matrix.test.ts',
// Standalone periodic self-gated A/B probe; template-literal testNames (auq-ab-${label}), no E2E map key — fail-open-safe, runs on every periodic sweep.
'test/skill-e2e-auq-verbose-vs-carved-ab.test.ts',
// bin-script pipeline test (spawns bun scripts, no model spend) that lives under the skill-e2e-* glob; no E2E map key exists for it.
'test/skill-e2e-memory-pipeline.test.ts',
]);
describe('E2E tier alignment (touchfiles declaration vs test self-gate)', () => {
+4 -4
View File
@@ -127,16 +127,16 @@ test('live periodic census fits the declared CI wall including setup', () => {
return paidShardWallUpperBoundMs(files, workers);
});
expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(periodicJob['timeout-minutes'] * 60_000);
expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(100);
expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(82);
const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount - 1);
expect(overlays).toHaveLength(6);
expect(overlays).toHaveLength(4);
expect(overlays.every(e => isOverlayTestFile(e.file))).toBe(true);
expect(m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount).map(e => e.file)).toEqual([AUTOPLAN_CHAIN_BUDGET.file]);
});
test('registered allocation is deterministic and preserves every discovered file', () => {
const files = collectPaidTestFiles();
expect(files).toHaveLength(119);
expect(files).toHaveLength(105);
const m = livePlan(files);
expect(livePlan([...files].reverse())).toEqual(m);
expect(m.entries.map(e => e.file).sort()).toEqual([...files].sort());
@@ -175,7 +175,7 @@ test('current detach supervision covers the live-census floor', () => {
const floor = Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05);
const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8'));
const configured = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]);
expect(floor).toBe(65541);
expect(floor).toBe(57330);
expect(configured).toBeGreaterThanOrEqual(floor);
expect(pkg.scripts['eval:bg:gate']).toContain('--timeout 36000');
});
+1 -1
View File
@@ -184,7 +184,7 @@ test('rejected completion does not erase a genuine earlier question or change un
test('completion evidence dependencies select exactly the seeded observation owners', () => {
const owners = ['plan-ceo-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-design-review-plan-mode',
'plan-devex-review-plan-mode', 'plan-mode-no-op', 'auto-decide-preserved', 'conductor-prose'].sort();
'plan-devex-review-plan-mode', 'plan-mode-no-op', 'auto-decide-preserved'].sort();
for (const file of ['test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json']) {
expect(selectTests([file], E2E_TOUCHFILES).selected.sort()).toEqual(owners);
}
-24
View File
@@ -1,24 +0,0 @@
{
"publicDecisionTail": "D1—ReducethepricingtiertoStripe-nativepieces,orbuildallfourcomponents?\r\rReplywithaletter(Conductorsession,sothisbriefisinprose).Project/branch:gstackedinburgh-v1,reviewingan\rexternal pricing-tier plan.\r\rELI10:Youwantmoredeveloperstosignup,soyou'readdingacheaperplan.Thedraftbuildsfournewthingstodo\rit: a Stripe price, a custm pricing web age,r own atabse tbletht remmbers who paid fowhat, ad a fast\rn-memory copy of tht table. Strpealready ships the table (Entitlements)and th page (rcing table). Every exra\rcopy of \"wh is allowed to dowhat\" is another place that can disagre withStripe when a card fails or awebhook\rarriveslate. The stakes: a developer who paidand sees \"upgrae required,\"or ono cnelled andkeeps access.\r\rStakesifwepickwrong:threesourcesoftruthforentitlementswithnoinvalidationdesignmeanssilentaccessbugs\rtht onlyshow upas angry support tickets.\r\rRecommendation:Abecauseitkeepseveryuser-visibleoutcomeofthedraftwhiledeletingthetwocomponentsmost\rlikely to causeentitlemnt drift,and it fits \"engineered enough\" an \"smallestiff that cleanly exressethe\rchange.\"\r\rCompleteness:A=9/10,B=7/10.Cisahold,notabuildoption,sonoscore.\r\rA)Stripe-nativeslimbuild(recommended).Completeness9/10.Human:~1week/CC:~1to2hours.AddoneStripePrice\rplus a Fature-to-Product mapping, mirror active_entitlement_sumryinto a singleexisting cstomer table hrough an\ridempotent webhok, gate fetures off that olumn, ship the pricing surfacebehinda feature flag (Stripe pricin\u001b[?25h",
"actualFlags": {
"proseAUQEverObserved": false,
"waitingEverObserved": true,
"scopeGateAutoSelectObserved": true
},
"owner": {
"sessionId": "4e820ffc-b6dd-47d7-b074-5c4b0a3cbb68",
"commandStartedAt": 1789041170391,
"nativePolledAt": 1789041383968,
"captureAt": "2026-09-10T11:56:33.483Z"
},
"provenance": {
"observation": ".context/ship-source-ao-delta-paid-20260910-v1/conductor-first-failure-ledger-v1/observation.json",
"observationSha256": "da8c8a342f44ae06bfdd092f1bdb6a395765553d65ac99eecaaa35dedb34545d",
"visible": ".context/ship-source-ao-delta-paid-20260910-v1/conductor-first-failure-ledger-v1/terminal.visible.log",
"visibleSha256": "2a2fca3e8d1b51694d76a3dbcfac524626f856e4ebf38f4bd093346edb64bf82",
"startCharacterOffset": 78469,
"publicTailSha256": "e551890b1d839cee7d7dce05120b8dc05a54f27fe8f69617aa99384c49316604",
"limit": "Exact public partial D1 tail; no completed native prose message retained. A later second option in tests is synthetic.",
"offsetUnit": "Unicode code points in decoded retained bytes, without newline normalization"
}
}
-2
View File
@@ -1,6 +1,4 @@
{
"command_reference": { "clarity": 4, "completeness": 3, "actionability": 4 },
"snapshot_flags": { "clarity": 4, "completeness": 4, "actionability": 4 },
"browse_skill": { "clarity": 4, "completeness": 4, "actionability": 4 },
"qa_workflow": { "clarity": 4, "completeness": 4, "actionability": 4 },
"qa_health_rubric": { "clarity": 4, "completeness": 3, "actionability": 4 }
-44
View File
@@ -305,50 +305,6 @@ export const OVERLAY_FIXTURES: OverlayFixture[] = [
pass: lowerIsBetter20Pct,
},
{
id: 'opus-4-7-effort-match-trivial-sonnet',
overlayPath: 'model-overlays/opus-4-7.md',
model: 'claude-sonnet-4-6',
trials: 10,
concurrency: 3,
direction: 'lower_is_better',
maxTurns: 8,
setupWorkspace: (dir) => {
fs.writeFileSync(
path.join(dir, 'config.json'),
'{"name": "demo", "version": "1.0.0"}\n',
);
},
userPrompt: "What's the version in config.json? Return only a JSON object with the version key and its exact string value. " +
"The final message is consumed directly by JSON.parse: return the exact JSON object, with no Markdown fences and no other prose.",
metric: reportedThinkingTokens,
metricName: 'reported_thinking_tokens',
verify: (r) => assertFinalJson(r, { version: '1.0.0' }),
comparison: { direction: 'lower_is_better', minimum: 0 },
pass: lowerIsBetter20Pct,
},
{
id: 'opus-4-7-literal-interpretation-sonnet',
overlayPath: 'model-overlays/opus-4-7.md',
model: 'claude-sonnet-4-6',
trials: 10,
concurrency: 3,
direction: 'higher_is_better',
allowedTools: ['Read', 'Glob', 'Grep', 'Bash', 'Edit', 'Write'],
maxTurns: 15,
setupWorkspace: setupLiteralWorkspace,
userPrompt: 'Fix the failing tests. Preserve the specified behavior and repair the implementation.',
metric: (_r, dir, deadlineAt) => {
if (!dir) throw new Error('literal fixture metric needs its workspace');
return correctLiteralTargets(dir, deadlineAt);
},
metricName: 'correct_target_behaviors',
taskCorrect: (metric) => metric === 3,
allowedChanges: ['src/auth.ts', 'src/billing.ts', 'src/notifications.ts'],
comparison: { direction: 'higher_is_better', minimum: 0, maximum: 3 },
pass: higherIsBetter20Pct,
},
];
// Validate at module load so a broken fixture fails fast at test startup,
-174
View File
@@ -1,174 +0,0 @@
/**
* Gemini CLI E2E smoke test — verify Gemini CLI can start and discover skills.
*
* This is a lightweight smoke test, not a full integration test. Gemini CLI
* gets lost in worktrees and times out on complex tasks. The smoke test
* validates that the skill files are structured correctly for Gemini's
* .agents/skills/ discovery mechanism.
*
* Prerequisites:
* - `gemini` binary installed (npm install -g @google/gemini-cli)
* - Gemini authenticated via ~/.gemini/ config or GEMINI_API_KEY env var
* - EVALS=1 env var set (same gate as Claude E2E tests)
*
* Skips gracefully when prerequisites are not met.
*/
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
import { JUDGE_MS } from './helpers/eval-budgets';
import { runGeminiSkill } from './helpers/gemini-session-runner';
import type { GeminiResult } from './helpers/gemini-session-runner';
import { EvalCollector } from './helpers/eval-store';
import { selectTests, detectBaseBranch, getChangedFiles, E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from './helpers/touchfiles';
import { createTestWorktree, harvestAndCleanup } from './helpers/e2e-helpers';
import * as path from 'path';
const ROOT = path.resolve(import.meta.dir, '..');
// --- Prerequisites check ---
const GEMINI_AVAILABLE = (() => {
try {
const result = Bun.spawnSync(['which', 'gemini'], { timeout: 30_000 });
return result.exitCode === 0;
} catch { return false; }
})();
// A binary on PATH is not enough: the CLI can be present but UNUSABLE — the
// individual code-assist auth path was deprecated upstream ("migrate to the
// Antigravity suite"), which fails every run before any model call, and flag
// churn (--skip-trust removed in 0.34) errors at argv parse. Probe with a
// bare --help: a CLI that can't even print usage is unusable, and a working
// one is cheap to confirm. Deeper auth failures are classified per-run.
const GEMINI_USABLE = GEMINI_AVAILABLE && (() => {
try {
const result = Bun.spawnSync(['gemini', '--help'], { timeout: 15_000 });
return result.exitCode === 0;
} catch { return false; }
})();
const evalsEnabled = !!process.env.EVALS;
// External-service tests are periodic-tier (CLAUDE.md tiering rule 3):
// "Requires external service (Codex, Gemini)? -> periodic". The positive
// form below is the canonical whole-file guard shape — the sharded runner's
// classifyPaidTestFile greps for it to exclude this file from gate.
const tierOk = process.env.EVALS_TIER === 'periodic';
// Skip all tests if gemini is not available/usable, EVALS is not set, or
// we're in the gate tier.
const SKIP = !GEMINI_USABLE || !evalsEnabled || !tierOk;
const describeGemini = SKIP ? describe.skip : describe;
// Log why we're skipping (helpful for debugging CI)
if (!evalsEnabled) {
// Silent — same as Claude E2E tests, EVALS=1 required
} else if (!tierOk) {
process.stderr.write('\nGemini E2E: SKIPPED — external-service test, periodic tier only (EVALS_TIER === \'periodic\')\n');
} else if (!GEMINI_AVAILABLE) {
process.stderr.write('\nGemini E2E: SKIPPED — gemini binary not found (install: npm i -g @google/gemini-cli)\n');
} else if (!GEMINI_USABLE) {
process.stderr.write('\nGemini E2E: SKIPPED — gemini CLI present but unusable (auth path deprecated upstream or CLI broken; try updating @google/gemini-cli)\n');
}
// --- Diff-based test selection ---
// Gemini E2E touchfiles — DERIVED from the canonical map, never a local fork
// (the old hand-copy kept a gitignored '.agents/skills/**' pattern that can
// never match a git diff and missed canonical deps — same drift class as the
// codex copy).
const GEMINI_E2E_TOUCHFILES: Record<string, string[]> = Object.fromEntries(
(['gemini-smoke'] as const).map((key) => {
if (!E2E_TOUCHFILES[key]) throw new Error(`canonical E2E_TOUCHFILES lost key '${key}' — fix the map, not this file`);
return [key, E2E_TOUCHFILES[key]];
}),
);
let selectedTests: string[] | null = null; // null = run all
if (evalsEnabled && !process.env.EVALS_ALL) {
const baseBranch = process.env.EVALS_BASE
|| detectBaseBranch(ROOT)
|| 'main';
const changedFiles = getChangedFiles(baseBranch, ROOT);
if (changedFiles.length > 0) {
const selection = selectTests(changedFiles, GEMINI_E2E_TOUCHFILES, GLOBAL_TOUCHFILES);
selectedTests = selection.selected;
process.stderr.write(`\nGemini E2E selection (${selection.reason}): ${selection.selected.length}/${Object.keys(GEMINI_E2E_TOUCHFILES).length} tests\n`);
if (selection.skipped.length > 0) {
process.stderr.write(` Skipped: ${selection.skipped.join(', ')}\n`);
}
process.stderr.write('\n');
}
}
/** Skip an individual test if not selected by diff-based selection. */
function testIfSelected(testName: string, fn: () => Promise<void>, timeout: number) {
const shouldRun = selectedTests === null || selectedTests.includes(testName);
(shouldRun ? test.concurrent : test.skip)(testName, fn, timeout);
}
// --- Eval result collector ---
const evalCollector = evalsEnabled && !SKIP ? new EvalCollector('e2e-gemini') : null;
function recordGeminiE2E(name: string, result: GeminiResult, passed: boolean) {
evalCollector?.addTest({
name,
suite: 'gemini-e2e',
tier: 'e2e',
passed,
duration_ms: result.durationMs,
cost_usd: 0,
output: result.output?.slice(0, 2000),
turns_used: result.toolCalls.length,
exit_reason: result.exitCode === 0 ? 'success' : `exit_code_${result.exitCode}`,
});
}
function logGeminiCost(label: string, result: GeminiResult) {
const durationSec = Math.round(result.durationMs / 1000);
console.log(`${label}: ${result.tokens} tokens, ${result.toolCalls.length} tool calls, ${durationSec}s`);
}
// Finalize eval results on exit
afterAll(async () => {
if (evalCollector) {
await evalCollector.finalize();
}
});
// --- Tests ---
describeGemini('Gemini E2E', () => {
let testWorktree: string;
beforeAll(() => {
testWorktree = createTestWorktree('gemini');
});
afterAll(() => {
harvestAndCleanup('gemini');
});
testIfSelected('gemini-smoke', async () => {
// Smoke test: can Gemini start, read the repo, and produce output?
// Uses a simple prompt that doesn't require skill invocation or complex navigation.
const result = await runGeminiSkill({
prompt: 'What is this project? Answer in one sentence based on the README.',
timeoutMs: JUDGE_MS,
cwd: testWorktree,
});
logGeminiCost('gemini-smoke', result);
// Pass if Gemini produced any meaningful output (even with non-zero exit from timeout)
const hasOutput = result.output.length > 10;
const passed = hasOutput;
recordGeminiE2E('gemini-smoke', result, passed);
expect(result.output.length, 'Gemini should produce output').toBeGreaterThan(10);
}, JUDGE_MS);
});
+34
View File
@@ -369,6 +369,40 @@ describe('gstack-skill-start behavior', () => {
const out = runStart();
expect(out).toMatch(/^ARTIFACTS_SYNC: off$/m);
});
test('artifacts-sync consent is asked before any artifacts egress, and only in interactive sessions', () => {
const gh = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ss-privacy-'));
const bin = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ss-privacy-bin-'));
const remote = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ss-privacy-remote-'));
const git = (args: string[], cwd: string) => execFileSync('git', args, { cwd, timeout: 30_000, stdio: 'pipe' });
try {
fs.writeFileSync(path.join(bin, 'gbrain'), '#!/bin/sh\nexit 0\n', { mode: 0o755 });
git(['init', '-q', '--bare'], remote);
git(['init', '-q', '-b', 'main'], gh);
git(['remote', 'add', 'origin', remote], gh);
const pullStamp = path.join(gh, '.brain-last-pull');
const start = (config: string, env: Record<string, string> = {}) => {
fs.writeFileSync(path.join(gh, 'config.yaml'), `update_check: false\n${config}`);
return runStart([], { GSTACK_HOME: gh, PATH: `${bin}${path.delimiter}${process.env.PATH}`, ...env });
};
const gates = (out: string) => (out.match(/^GSTACK_INSTRUCTION_BEGIN: privacy-stop-gate/gm) ?? []).length;
const pending = start('');
expect(gates(pending)).toBe(1);
expect(pending).toContain('How much should sync?');
expect(pending).toMatch(/^ARTIFACTS_SYNC: off$/m);
expect(fs.existsSync(pullStamp)).toBe(false);
expect(gates(start('', { GSTACK_SESSION_KIND: 'spawned' }))).toBe(0);
expect(fs.existsSync(pullStamp)).toBe(false);
const consented = start('artifacts_sync_mode: full\nartifacts_sync_mode_prompted: true\n');
expect(gates(consented)).toBe(0);
expect(fs.existsSync(pullStamp)).toBe(true);
} finally {
for (const dir of [gh, bin, remote]) fs.rmSync(dir, { recursive: true, force: true });
}
});
});
describe('gstack-skill-end', () => {
+2 -3
View File
@@ -106,9 +106,8 @@ export const FILE_RETRY_BUDGETS = [
...STRICT_RETRY_CASE_BUDGETS,
...[
// Sixteen workflow judges include their 10s recording grace; the other
// eleven judges retain 120s. Supervise all 27 and the existing one retry.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 16 * (JUDGE_MS + 10_000) + 11 * JUDGE_MS, retries: 1 },
{ file: 'test/codex-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_LONG_MS + 10_000), retries: 1 },
// seven judges retain 120s. Supervise all 23 and the existing one retry.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 16 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, retries: 1 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, retries: 1 },
-104
View File
@@ -1,104 +0,0 @@
import { describe, test, expect } from 'bun:test';
import { parseGeminiJSONL } from './gemini-session-runner';
// Fixture: actual Gemini CLI stream-json output with tool use
const FIXTURE_LINES = [
'{"type":"init","timestamp":"2026-03-20T15:14:46.455Z","session_id":"test-session-123","model":"auto-gemini-3"}',
'{"type":"message","timestamp":"2026-03-20T15:14:46.456Z","role":"user","content":"list the files"}',
'{"type":"message","timestamp":"2026-03-20T15:14:49.650Z","role":"assistant","content":"I will list the files.","delta":true}',
'{"type":"tool_use","timestamp":"2026-03-20T15:14:49.690Z","tool_name":"run_shell_command","tool_id":"cmd_1","parameters":{"command":"ls"}}',
'{"type":"tool_result","timestamp":"2026-03-20T15:14:49.931Z","tool_id":"cmd_1","status":"success","output":"file1.ts\\nfile2.ts"}',
'{"type":"message","timestamp":"2026-03-20T15:14:51.945Z","role":"assistant","content":"Here are the files.","delta":true}',
'{"type":"result","timestamp":"2026-03-20T15:14:52.030Z","status":"success","stats":{"total_tokens":27147,"input_tokens":26928,"output_tokens":87,"cached":0,"duration_ms":5575,"tool_calls":1}}',
];
describe('parseGeminiJSONL', () => {
test('extracts session ID from init event', () => {
const parsed = parseGeminiJSONL(FIXTURE_LINES);
expect(parsed.sessionId).toBe('test-session-123');
});
test('concatenates assistant message deltas into output', () => {
const parsed = parseGeminiJSONL(FIXTURE_LINES);
expect(parsed.output).toBe('I will list the files.Here are the files.');
});
test('ignores user messages', () => {
const lines = [
'{"type":"message","role":"user","content":"this should be ignored"}',
'{"type":"message","role":"assistant","content":"this should be kept","delta":true}',
];
const parsed = parseGeminiJSONL(lines);
expect(parsed.output).toBe('this should be kept');
});
test('extracts tool names from tool_use events', () => {
const parsed = parseGeminiJSONL(FIXTURE_LINES);
expect(parsed.toolCalls).toHaveLength(1);
expect(parsed.toolCalls[0]).toBe('run_shell_command');
});
test('extracts total tokens from result stats', () => {
const parsed = parseGeminiJSONL(FIXTURE_LINES);
expect(parsed.tokens).toBe(27147);
});
test('skips malformed lines without throwing', () => {
const lines = [
'{"type":"init","session_id":"ok"}',
'this is not json',
'{"type":"message","role":"assistant","content":"hello","delta":true}',
'{incomplete json',
'{"type":"result","status":"success","stats":{"total_tokens":100}}',
];
const parsed = parseGeminiJSONL(lines);
expect(parsed.sessionId).toBe('ok');
expect(parsed.output).toBe('hello');
expect(parsed.tokens).toBe(100);
});
test('skips empty and whitespace-only lines', () => {
const lines = [
'',
' ',
'{"type":"init","session_id":"s1"}',
'\t',
'{"type":"result","status":"success","stats":{"total_tokens":50}}',
];
const parsed = parseGeminiJSONL(lines);
expect(parsed.sessionId).toBe('s1');
expect(parsed.tokens).toBe(50);
});
test('handles empty input', () => {
const parsed = parseGeminiJSONL([]);
expect(parsed.output).toBe('');
expect(parsed.toolCalls).toHaveLength(0);
expect(parsed.tokens).toBe(0);
expect(parsed.sessionId).toBeNull();
});
test('handles missing fields gracefully', () => {
const lines = [
'{"type":"init"}', // no session_id
'{"type":"message","role":"assistant"}', // no content
'{"type":"tool_use"}', // no tool_name
'{"type":"result","status":"success"}', // no stats
];
const parsed = parseGeminiJSONL(lines);
expect(parsed.sessionId).toBeNull();
expect(parsed.output).toBe('');
expect(parsed.toolCalls).toHaveLength(0);
expect(parsed.tokens).toBe(0);
});
test('handles multiple tool_use events', () => {
const lines = [
'{"type":"tool_use","tool_name":"run_shell_command","tool_id":"cmd_1","parameters":{"command":"ls"}}',
'{"type":"tool_use","tool_name":"read_file","tool_id":"cmd_2","parameters":{"path":"foo.ts"}}',
'{"type":"tool_use","tool_name":"run_shell_command","tool_id":"cmd_3","parameters":{"command":"cat bar.ts"}}',
];
const parsed = parseGeminiJSONL(lines);
expect(parsed.toolCalls).toEqual(['run_shell_command', 'read_file', 'run_shell_command']);
});
});
-262
View File
@@ -1,262 +0,0 @@
/**
* Gemini CLI subprocess runner for skill E2E testing.
*
* Spawns `gemini -p` as an independent process, parses its stream-json
* output, and returns structured results. Follows the same pattern as
* codex-session-runner.ts but adapted for the Gemini CLI.
*
* Key differences from Codex session-runner:
* - Uses `gemini -p` instead of `codex exec`
* - Output is NDJSON with event types: init, message, tool_use, tool_result, result
* - Uses `--output-format stream-json --yolo` instead of `--json -s read-only`
* (`--skip-trust` was removed in gemini-cli 0.34; folder trust is settings-driven now)
* - No temp HOME needed — Gemini discovers skills from `.agents/skills/` in cwd
* - Message events are streamed with `delta: true` — must concatenate
*/
import * as path from 'path';
import { spawn } from 'child_process';
import { Readable } from 'node:stream';
import { hermeticChildEnv } from './hermetic-env';
import { killProcessGroup } from '../../scripts/test-strict-output';
// --- Interfaces ---
export interface GeminiResult {
output: string; // Full assistant message text (concatenated deltas)
toolCalls: string[]; // Tool names from tool_use events
tokens: number; // Total tokens used
exitCode: number; // Process exit code
durationMs: number; // Wall clock time
sessionId: string | null; // Session ID from init event
rawLines: string[]; // Raw JSONL lines for debugging
}
// --- JSONL parser ---
export interface ParsedGeminiJSONL {
output: string;
toolCalls: string[];
tokens: number;
sessionId: string | null;
}
/**
* Parse an array of JSONL lines from `gemini -p --output-format stream-json`.
* Pure function — no I/O, no side effects.
*
* Handles these Gemini event types:
* - init → extract session_id
* - message (role=assistant, delta=true) → concatenate content into output
* - tool_use → extract tool_name
* - tool_result → logged but not extracted
* - result → extract token usage from stats
*/
export function parseGeminiJSONL(lines: string[]): ParsedGeminiJSONL {
const outputParts: string[] = [];
const toolCalls: string[] = [];
let tokens = 0;
let sessionId: string | null = null;
for (const line of lines) {
if (!line.trim()) continue;
try {
const obj = JSON.parse(line);
const t = obj.type || '';
if (t === 'init') {
const sid = obj.session_id || '';
if (sid) sessionId = sid;
} else if (t === 'message') {
if (obj.role === 'assistant' && obj.content) {
outputParts.push(obj.content);
}
} else if (t === 'tool_use') {
const name = obj.tool_name || '';
if (name) toolCalls.push(name);
} else if (t === 'result') {
const stats = obj.stats || {};
tokens = (stats.total_tokens || 0);
}
} catch { /* skip malformed lines */ }
}
return {
output: outputParts.join(''),
toolCalls,
tokens,
sessionId,
};
}
// --- Main runner ---
/**
* Run a prompt via `gemini -p` and return structured results.
*
* Spawns gemini with stream-json output, parses JSONL events,
* and returns a GeminiResult. Skips gracefully if gemini binary is not found.
*/
export async function runGeminiSkill(opts: {
prompt: string; // What to ask Gemini
timeoutMs?: number; // Default 300000 (5 min)
cwd?: string; // Working directory (where .agents/skills/ lives)
}): Promise<GeminiResult> {
const {
prompt,
timeoutMs = 300_000,
cwd,
} = opts;
const startTime = Date.now();
// Check if gemini binary exists
const whichResult = Bun.spawnSync(['which', 'gemini'], { timeout: 30_000 });
if (whichResult.exitCode !== 0) {
return {
output: 'SKIP: gemini binary not found',
toolCalls: [],
tokens: 0,
exitCode: -1,
durationMs: Date.now() - startTime,
sessionId: null,
rawLines: [],
};
}
// Build gemini command.
// --skip-trust was REMOVED in gemini-cli 0.34 ("Unknown arguments:
// skip-trust"); folder trust moved to settings and no longer needs a flag
// for headless runs. --yolo still auto-approves tool actions.
const args = ['-p', prompt, '--output-format', 'stream-json', '--yolo'];
// Spawn gemini — uses real HOME for auth (~/.gemini; HOME is allowlisted),
// cwd for skill discovery. Hermetic scrub with gemini's auth surface
// re-admitted (previously this spawn inherited the full operator env).
// node:child_process spawn with `detached` (own process group) — mirrors
// session-runner.ts. A bare kill signalled only gemini itself; tool
// subprocesses survived as orphans holding our pipes open (the same
// blocked-drain hang the claude runner fixed — this copy lacked it).
const proc = spawn('gemini', args, {
cwd: cwd || process.cwd(),
stdio: ['ignore', 'pipe', 'pipe'],
detached: process.platform !== 'win32',
env: hermeticChildEnv(undefined, {
extraAllow: ['GEMINI_API_KEY', 'GOOGLE_API_KEY', 'GOOGLE_APPLICATION_CREDENTIALS', 'GOOGLE_CLOUD_*', 'GEMINI_*'],
}),
});
const stdoutWeb = Readable.toWeb(proc.stdout!) as ReadableStream<Uint8Array>;
const stderrWeb = Readable.toWeb(proc.stderr!) as ReadableStream<Uint8Array>;
const procExited: Promise<number> = new Promise((resolve) => {
proc.on('close', (code) => resolve(code ?? 1));
proc.on('error', () => resolve(1));
});
// Race against timeout
let timedOut = false;
const timeoutId = setTimeout(() => {
timedOut = true;
// Group SIGKILL + reader cancel: kill the whole tree AND unblock the
// read loop even if a stray grandchild survives the group kill.
killProcessGroup(proc, 'SIGKILL');
reader.cancel().catch(() => { /* stream already closed */ });
}, timeoutMs);
// Stream and collect JSONL from stdout
const collectedLines: string[] = [];
const stderrPromise = new Response(stderrWeb).text();
const reader = stdoutWeb.getReader();
const decoder = new TextDecoder();
let buf = '';
try {
while (true) {
const { done, value } = await reader.read();
if (done) break;
buf += decoder.decode(value, { stream: true });
const lines = buf.split('\n');
buf = lines.pop() || '';
for (const line of lines) {
if (!line.trim()) continue;
collectedLines.push(line);
// Real-time progress to stderr
try {
const event = JSON.parse(line);
if (event.type === 'tool_use' && event.tool_name) {
const elapsed = Math.round((Date.now() - startTime) / 1000);
process.stderr.write(` [gemini ${elapsed}s] tool: ${event.tool_name}\n`);
} else if (event.type === 'message' && event.role === 'assistant' && event.content) {
const elapsed = Math.round((Date.now() - startTime) / 1000);
process.stderr.write(` [gemini ${elapsed}s] message: ${event.content.slice(0, 100)}\n`);
}
} catch { /* skip — parseGeminiJSONL will handle it later */ }
}
}
} catch { /* stream read error — fall through to exit code handling */ }
// Flush remaining buffer
if (buf.trim()) {
collectedLines.push(buf);
}
// Same orphan hazard as stdout: a grandchild holding stderr open would
// block this drain forever. Race against child exit + a short grace window
// (ported from session-runner.ts — the gemini copy lacked it).
const stderr = await Promise.race([
stderrPromise,
(async () => {
await procExited;
await new Promise((r) => setTimeout(r, 5_000));
return '';
})(),
]);
const exitCode = await procExited;
clearTimeout(timeoutId);
const durationMs = Date.now() - startTime;
// Parse all collected JSONL lines
const parsed = parseGeminiJSONL(collectedLines);
// Log stderr if non-empty (may contain auth errors, etc.)
if (stderr.trim()) {
process.stderr.write(` [gemini stderr] ${stderr.trim().slice(0, 200)}\n`);
}
// Environment-unusable classification: these are Google-side conditions no
// test assertion can act on — the deprecated individual code-assist auth
// path ("migrate to the Antigravity suite") and argv drift on older/newer
// CLIs. Return the same SKIP shape as binary-not-found so callers report
// SKIPPED instead of a false FAIL.
const unusableMarkers = [
'no longer supported for Gemini Code Assist',
'antigravity',
'Unknown arguments: skip-trust',
];
if (exitCode !== 0 && parsed.tokens === 0) {
const marker = unusableMarkers.find((m) => stderr.toLowerCase().includes(m.toLowerCase()));
if (marker) {
return {
output: `SKIP: gemini CLI unusable (${marker})`,
toolCalls: [],
tokens: 0,
exitCode: -1,
durationMs,
sessionId: null,
rawLines: collectedLines,
};
}
}
return {
output: parsed.output,
toolCalls: parsed.toolCalls,
tokens: parsed.tokens,
exitCode: timedOut ? 124 : exitCode,
durationMs,
sessionId: parsed.sessionId,
rawLines: collectedLines,
};
}
-2
View File
@@ -21,6 +21,4 @@ export const OVERLAY_CASE_FILES: Record<string, string> = {
'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts': 'opus-4-7-effort-match-trivial',
'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts': 'opus-4-7-literal-interpretation',
'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts': 'claude-dedicated-tools-vs-bash-sonnet',
'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial-sonnet.test.ts': 'opus-4-7-effort-match-trivial-sonnet',
'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation-sonnet.test.ts': 'opus-4-7-literal-interpretation-sonnet',
};
+2 -3
View File
@@ -11,8 +11,8 @@ import { matchGlob } from './touchfiles';
/** The exact globs package.json's `test:gate` passes to `bun test`. */
export const PAID_TEST_GLOBS = [
// skill-llm-eval* (not just the base file): skill-llm-eval-spec.test.ts
// fell outside the exact glob and could never run in any lane.
// skill-llm-eval* (not just the base file): a sibling judge file outside
// the exact glob could never run in any lane.
'test/skill-llm-eval*.test.ts',
'test/skill-e2e-*.test.ts',
'test/skill-routing-e2e.test.ts',
@@ -22,7 +22,6 @@ export const PAID_TEST_GLOBS = [
// entered the paid census. The same bug class as the deleted pre-split
// monolith (see test/paid-shards.test.ts's regression pin).
'test/codex-e2e*.test.ts',
'test/gemini-e2e.test.ts',
'test/llm-judge-recommendation.test.ts',
'test/carve-section-loading*.test.ts',
] as const;
+25 -13
View File
@@ -15,20 +15,32 @@
* the re-entry mechanism.
*/
export const PERIODIC_CI_EXCLUDE: Record<string, { reason: string; tracking: string }> = {
'test/skill-e2e-ship-idempotency.test.ts': {
reason:
'documented-red: the PTY child sits at the Claude Code welcome screen for the full budget '
+ '(readiness/typing race vs CLI 2.1.x); never green since it was born in v1.63',
tracking: 'TODOS.md "periodic tier — three documented-red tests need structural repair" (1 of 3 resolved: sidebar trio already deleted)',
'test/codex-e2e.test.ts': {
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/skill-e2e-brain-privacy-gate.test.ts': {
reason:
'documented-red: the artifacts-sync stop-gate preconditions do not survive the hermetic env '
+ 'even with per-test HOME/GSTACK_HOME injection; never green anywhere',
tracking: 'TODOS.md "periodic tier — three documented-red tests need structural repair"',
'test/codex-e2e-sol-scope.test.ts': {
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/skill-e2e-ios.test.ts': {
reason: 'requires a live iOS device/simulator toolchain (xcodebuild, devicectl) — manual hardware, not a CI runner capability',
tracking: 'TODOS.md "skill-e2e-ios CI story" (device/runner decision)',
'test/codex-e2e-shared-libs.test.ts': {
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/codex-e2e-recommendation-substance.test.ts': {
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/skill-e2e-outside-voice.test.ts': {
reason: 'needs both the claude and codex CLIs; codex is not in the CI image, so every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/skill-e2e-aside.test.ts': {
reason: 'needs macOS with the Aside app open (asideAvailable()); CI runners are Linux, so every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/skill-e2e-ios-device.test.ts': {
reason: 'needs a physical iPhone over USB/devicectl — manual hardware, not a CI runner capability',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
};
+2 -2
View File
@@ -3,8 +3,8 @@ import * as path from 'node:path';
/**
* Provider adapter interface — uniform contract for Claude, GPT, Gemini.
*
* Each adapter wraps an existing runner (session-runner.ts, codex-session-runner.ts,
* gemini-session-runner.ts) and normalizes its per-provider result shape into the
* Each adapter wraps an existing runner or CLI (session-runner.ts,
* codex-session-runner.ts, the gemini CLI) and normalizes its per-provider result shape into the
* RunResult below. The benchmark harness only talks to adapters through this
* interface, never to the underlying runners directly.
*/
+11 -80
View File
@@ -334,23 +334,6 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
// Conductor → prose decision brief (Conductor signal makes prose the default;
// the PreToolUse hook denies the flaky tool). Touches the resolver that owns
// the Conductor rule, the preamble signal, the hook, and the detection helper.
'conductor-prose': [
'lib/claude-public-transcript.ts',
'test/auto-decide-recommendation-scope.test.ts',
'test/fixtures/auto-decide-recommendation-361c.json',
'test/auto-decide-target-identity.test.ts',
'test/fixtures/auto-decide-target-361c.json',
'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json',
'test/helpers/plan-count-transcript.ts', 'test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures/plan-count-cross-cwd-ancestry-0bcd.json', 'test/plan-count-session-cwd.test.ts',
'scripts/resolvers/learnings.ts',
"test/plan-scope-recovery-av.test.ts",
"test/fixtures/plan-scope-recovery-av.json", 'test/pty-screen-unicode-ap.test.ts', 'test/eng-scope-entry-ap.test.ts',
'test/conductor-prose-observation-ao.test.ts', 'test/fixtures/conductor-prose-ao.json',
'test/auto-decide-saved-ai.test.ts', 'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'bin/gstack-session-kind', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-preamble-bash.ts', 'scripts/resolvers/preamble.ts', 'plan-eng-review/**', 'hosts/claude/hooks/question-preference-hook.ts', 'hosts/claude/hooks/spawned-directive.ts', 'lib/is-conductor.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-conductor-prose.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**',
"test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-completion-status.ts",
'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/helpers/owned-claude-transcript.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/testing.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts'
],
// Native question capture and interactive workflow probes.
'auq-format-gate': ['test/session-runner-stream-lifecycle.test.ts', 'plan-ceo-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completeness-section.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/session-runner.ts', 'test/helpers/llm-judge.ts', 'test/skill-e2e-ask-user-question-format-compliance.test.ts',
@@ -393,9 +376,6 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
"test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts",
'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/plan-design-with-ui-fixture.test.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/helpers/plan-review-cases.ts', 'test/helpers/plan-review-board-feedback.ts', 'test/plan-review-board-feedback.test.ts', 'test/fixtures/design-board-questions.json', 'test/fixtures/design-outside-voices-question.json', 'design/src/daemon-state.ts', 'design/src/daemon.ts', 'design/test/daemon-tests-fixtures.ts', 'design/src/daemon-client.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'bin/gstack-paths', 'bin/gstack-slug', 'scripts/resolvers/design.ts'
],
'ship-idempotency-pty': ['ship/**', 'bin/gstack-next-version', 'bin/gstack-version-bump', 'scripts/resolvers/sections.ts', 'lib/worktree.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-ship-idempotency.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt',
'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/testing.ts'
],
'tpa-present': ['test/session-runner-stream-lifecycle.test.ts', 'scripts/resolvers/third-party-actions.ts', 'ship/SKILL.md.tmpl', 'ship/sections/apple-release.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-third-party-actions.test.ts', 'test/helpers/third-party-actions.ts',
'test/third-party-actions-recording.test.ts', 'lib/eval-model.ts'
],
@@ -784,9 +764,6 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
'test/fixtures/plan-count-quoted-frame-ak.json',
'docs/askuserquestion-split.md', 'test/resolver-ask-user-format.test.ts', 'bin/gstack-slug', 'test/helpers/hermetic-env.test.ts', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/ceo-finding-fixture.test.ts', 'test/helpers/owned-claude-transcript.ts', 'test/eval-budgets-policy.test.ts', 'test/fixtures/webfetch-permission.json', 'test/plan-skill-webfetch-permission.test.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'test/helpers/plan-review-decisions.ts', 'test/plan-review-decisions.test.ts', 'test/helpers/plan-review-cases.ts', 'test/plan-review-cases.test.ts', 'test/helpers/llm-judge.ts', 'lib/eval-model.ts', 'test/skill-e2e-plan-decision-classification.test.ts', 'test/fixtures/plan-decision-classification.ts', 'test/plan-review-calibration.test.ts', 'test/helpers/ceo-split-question-policy.ts', 'test/fixtures/ceo-split-actor-6aef.json', 'test/helpers/ceo-mode-option.ts', 'test/ceo-mode-option.test.ts', 'test/ceo-split-collection.test.ts', 'test/fixtures/ceo-split-collection-0bcd.json', 'test/ceo-split-question-policy.test.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/tasks-section.ts'
],
'brain-privacy-gate': ['bin/gstack-skill-start', 'bin/gstack-skill-end', 'scripts/resolvers/preamble/generate-brain-sync-block.ts', 'scripts/resolvers/preamble.ts', 'bin/gstack-brain-sync', 'bin/gstack-artifacts-init', 'bin/gstack-config', 'test/helpers/agent-sdk-runner.ts', 'test/skill-e2e-brain-privacy-gate.test.ts',
'test/agent-sdk-runner.test.ts'
],
// /setup-gbrain Path 4 (Remote MCP) — happy + bad-token end-to-end via
// Agent SDK. Gate-tier (deterministic stub server, fixed inputs); fires
@@ -864,11 +841,6 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
'plan-tune-inspect': ['test/session-runner-stream-lifecycle.test.ts', 'plan-tune/**', 'scripts/question-registry.ts', 'scripts/psychographic-signals.ts', 'scripts/one-way-doors.ts', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'bin/gstack-developer-profile', 'test/skill-e2e-plan-tune.test.ts'],
// /plan-tune cathedral (T16 — 5 E2E scenarios, all gate per D12)
'plan-tune-hook-capture': ['hosts/claude/hooks/**', 'bin/gstack-question-log', 'bin/gstack-developer-profile', 'plan-tune/**', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'],
'plan-tune-enforcement': ['hosts/claude/hooks/**', 'bin/gstack-question-preference', 'scripts/question-registry.ts', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'],
'plan-tune-annotation': ['hosts/claude/hooks/**', 'scripts/declared-annotation.ts', 'scripts/psychographic-signals.ts', 'scripts/question-registry.ts', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'],
'plan-tune-codex-import': ['bin/gstack-codex-session-import', 'bin/gstack-question-log', 'docs/spikes/codex-session-format.md', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'],
'plan-tune-dream-cycle': ['bin/gstack-distill-free-text', 'bin/gstack-distill-apply', 'hosts/claude/hooks/**', 'plan-tune/**', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'],
// Codex offering verification
'codex-offered-office-hours': ['test/session-runner-stream-lifecycle.test.ts', 'test/paid-retry-supervision.test.ts', 'office-hours/**', 'scripts/gen-skill-docs.ts', 'test/skill-e2e-plan.test.ts',
@@ -1013,7 +985,6 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
],
// Gemini E2E — smoke test only (Gemini gets lost in worktrees on complex tasks)
'gemini-smoke': ['scripts/gen-skill-docs.ts', 'test/helpers/gemini-session-runner.ts', 'lib/worktree.ts', 'test/gemini-e2e.test.ts'],
// Coverage audit (shared fixture) + triage + gates
@@ -1264,15 +1235,16 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
"test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts",
],
'journey-negatives': ['test/session-runner-stream-lifecycle.test.ts',
'test/skill-fixture.test.ts',
"test/plan-scope-recovery-av.test.ts",
"test/fixtures/plan-scope-recovery-av.json",
"test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts',
"test/design-scope-entry-aq.test.ts",
"test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts",
],
// Opus 4.7 behavior evals — keys match testName: values in the test file.
// Routing sub-tests use template literal `routing-${c.name}` testNames,
// which the touchfile completeness scanner skips; they inherit selection
// from the file-level touchfile entry via GLOBAL_TOUCHFILES.
'fanout-arm-overlay-on':
['test/session-runner-stream-lifecycle.test.ts', 'model-overlays/claude.md', 'model-overlays/opus-4-7.md', 'scripts/models.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-opus-47.test.ts'],
'fanout-arm-overlay-off':
['test/session-runner-stream-lifecycle.test.ts', 'model-overlays/claude.md', 'model-overlays/opus-4-7.md', 'scripts/models.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-opus-47.test.ts'],
// Overlay efficacy harness (SDK) — measures whether overlay nudges change
// behavior under @anthropic-ai/claude-agent-sdk (closer to real Claude Code
@@ -1280,22 +1252,10 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
// completeness scanner doesn't require them; these entries exist for
// diff-based selection accuracy.
// /ios-qa — agent flow E2E. Daemon + stub StateServer + codegen
// exercised end-to-end. The no-device path is gate-tier; the with-device
// path requires GSTACK_HAS_IOS_DEVICE=1 and is periodic-tier.
'ios-qa-e2e': ['ios-qa/**', 'ios-fix/**', 'ios-design-review/**', 'ios-clean/**', 'ios-sync/**', 'test/skill-e2e-ios.test.ts'],
// Swift-build invariant test — requires the Swift toolchain. Compiles the
// fixture SPM package + runs the XCTest suite that validates the real
// Swift StateServer implementation (loopback bind, boot token rotation,
// session lock). Periodic-tier — Swift build is heavier than TS unit tests.
'ios-qa-swift-build': ['ios-qa/templates/**', 'test/fixtures/ios-qa/FixtureApp/**', 'test/skill-e2e-ios-swift-build.test.ts'],
// Real-device path — only runs with GSTACK_HAS_IOS_DEVICE=1 + a paired
// iPhone. Validates the CoreDevice agent + iOS SDK toolchain. Periodic-tier.
'ios-qa-device': ['ios-qa/templates/**', 'test/fixtures/ios-qa/FixtureApp/**', 'test/skill-e2e-ios-device.test.ts'],
// /spec end-to-end via PTY — exercises the full Phase 1→5 pipeline
// including --execute spawn. Periodic-tier — paid + non-deterministic.
'spec-execute': ['spec/**', 'scripts/resolvers/redact-doc.ts', 'scripts/resolvers/outside-voice.ts', 'scripts/resolvers/constants.ts', 'lib/outside-review-result.ts', 'bin/gstack-redact', 'lib/redact-engine.ts', 'test/skill-e2e-spec-execute.test.ts'],
// /office-hours brain-writeback path under fake gbrain CLI (v1.50.0.0
// T7). Drives /office-hours with a regenerated SKILL.md that has the
@@ -1370,16 +1330,10 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
'test/fixtures/eng-omitted-select-361c.json',
'test/skill-e2e-plan-decision-classification.test.ts', 'test/fixtures/plan-decision-classification.ts', 'test/plan-review-calibration.test.ts', 'test/helpers/plan-review-decisions.ts', 'test/plan-review-decisions.test.ts', 'test/helpers/plan-review-cases.ts', 'test/plan-review-cases.test.ts', 'test/helpers/llm-judge.ts', 'lib/eval-model.ts', 'test/helpers/e2e-helpers.ts', 'test/helpers/eval-store.ts', 'test/helpers/eval-budgets.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'docs/askuserquestion-split.md', 'plan-ceo-review/SKILL.md.tmpl', 'plan-ceo-review/sections/review-sections.md.tmpl', 'scripts/resolvers/tasks-section.ts'],
'health-reporting': ['health/**', 'test/skill-e2e-health.test.ts', 'test/helpers/health-eval-fixture.ts'],
'codex-plan-ceo-format-mode': ['test/paid-retry-supervision.test.ts', 'plan-ceo-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts'],
'codex-plan-ceo-format-approach': ['test/paid-retry-supervision.test.ts', 'plan-ceo-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts'],
'codex-plan-eng-format-coverage': ['test/paid-retry-supervision.test.ts', 'scripts/resolvers/learnings.ts', 'test/review-entry-and-design-clarity-au.test.ts', 'test/fixtures/plan-scope-recovery-av.json', 'test/plan-scope-recovery-av.test.ts', 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts', 'test/plan-review-cases.test.ts'],
'codex-plan-eng-format-kind': ['test/paid-retry-supervision.test.ts', 'scripts/resolvers/learnings.ts', 'test/review-entry-and-design-clarity-au.test.ts', 'test/fixtures/plan-scope-recovery-av.json', 'test/plan-scope-recovery-av.test.ts', 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts', 'test/plan-review-cases.test.ts'],
'overlay-harness-claude-dedicated-tools-vs-bash': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
'overlay-harness-opus-4-7-effort-match-trivial': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
'overlay-harness-opus-4-7-literal-interpretation': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
'overlay-harness-opus-4-7-effort-match-trivial-sonnet': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial-sonnet.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
'overlay-harness-opus-4-7-literal-interpretation-sonnet': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation-sonnet.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
};
/**
@@ -1503,7 +1457,6 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
// v1.21+ auto-mode regression tests
'office-hours-auto-mode': 'gate',
'auto-decide-preserved': 'periodic',
'conductor-prose': 'periodic',
// Real-PTY E2E batch — tier classification:
// gate: cheap, deterministic, run on every PR
@@ -1511,7 +1464,6 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
'auq-format-gate': 'gate', // ~$0.50/run, native SDK question capture, single skill probe
'plan-ceo-mode-routing': 'periodic', // ~$3/run, deep navigation through 8-12 prior AskUserQuestions
'plan-design-with-ui-scope': 'gate', // ~$0.80/run
'ship-idempotency-pty': 'periodic', // ~$3/run, real /ship in plan mode
'tpa-present': 'gate', // consent/credential safety guardrail; deterministic shims + grep asserts
'tpa-absent-linux': 'gate', // consent/credential safety guardrail; deterministic shims + grep asserts
'tpa-broken': 'gate', // consent/credential safety guardrail; deterministic shims + grep asserts
@@ -1540,7 +1492,6 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
// Privacy gate for gstack-brain-sync — periodic (non-deterministic LLM call,
// costs ~$0.30-$0.50 per run, not needed on every commit)
'brain-privacy-gate': 'periodic',
// /setup-gbrain Path 4 (Remote MCP) — periodic-tier. The stub HTTP
// server is deterministic but the model's interpretation of "follow
@@ -1578,11 +1529,6 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
'plan-tune-inspect': 'gate',
// /plan-tune cathedral (T16 per D12 — all gate)
'plan-tune-hook-capture': 'gate',
'plan-tune-enforcement': 'gate',
'plan-tune-annotation': 'gate',
'plan-tune-codex-import': 'gate',
'plan-tune-dream-cycle': 'gate',
// Codex offering verification
'codex-offered-office-hours': 'gate',
@@ -1646,7 +1592,6 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
'outside-voice-claude-code-to-codex': 'periodic',
'outside-plan-disabled-no-fallback': 'periodic',
'codex-sol-scope-termination': 'periodic',
'gemini-smoke': 'periodic',
// Design — gate for cheap functional, periodic for Opus/quality
'design-consultation-core': 'periodic',
@@ -1706,10 +1651,9 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
'journey-retro': 'periodic',
'journey-design-system': 'periodic',
'journey-visual-qa': 'periodic',
'journey-negatives': 'periodic',
// Opus 4.7 overlay evals — periodic (non-deterministic LLM behavior + Opus cost)
'fanout-arm-overlay-on': 'periodic',
'fanout-arm-overlay-off': 'periodic',
// Overlay efficacy harness (SDK, paid) — periodic only
@@ -1720,13 +1664,10 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
// on every Linux PR. Periodic keeps it in the weekly census on capable
// hosts; re-promote if a macOS runner lands (flagged decision in the
// test-infra overhaul plan).
'ios-qa-e2e': 'periodic',
// Swift toolchain only, no device required, but heavier than TS unit tests.
'ios-qa-swift-build': 'periodic',
// Requires a real connected + paired iPhone. Manual-trigger only.
'ios-qa-device': 'periodic',
// /spec end-to-end PTY pipeline (paid, non-deterministic — periodic-tier).
'spec-execute': 'periodic',
// WS2 arm benchmark — periodic: full build-shaped agentic workflows, paid,
// non-deterministic by construction (research instrument, not a gate).
@@ -1737,16 +1678,10 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
'plan-decision-classification': 'periodic',
'plan-devex-peer-comparison-classification': 'periodic',
'health-reporting': 'periodic',
'codex-plan-ceo-format-mode': 'periodic',
'codex-plan-ceo-format-approach': 'periodic',
'codex-plan-eng-format-coverage': 'periodic',
'codex-plan-eng-format-kind': 'periodic',
'overlay-harness-claude-dedicated-tools-vs-bash': 'periodic',
'overlay-harness-opus-4-7-effort-match-trivial': 'periodic',
'overlay-harness-opus-4-7-literal-interpretation': 'periodic',
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'periodic',
'overlay-harness-opus-4-7-effort-match-trivial-sonnet': 'periodic',
'overlay-harness-opus-4-7-literal-interpretation-sonnet': 'periodic',
};
/**
@@ -1754,16 +1689,12 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
*/
export const LLM_JUDGE_TOUCHFILES: Record<string, string[]> = {
'setup-browser-cookies/SKILL.md workflow': ['setup-browser-cookies/SKILL.md.tmpl', 'setup-browser-cookies/SKILL.md', 'BROWSER.md', 'test/helpers/cookie-workflow-judge-input.ts', 'test/cookie-workflow-judge-input.test.ts', 'test/helpers/cookie-workflow-manual-review.ts', 'test/cookie-workflow-manual-review.test.ts', 'test/helpers/manual-judge-review-fixture.ts', '.github/cookie-workflow-manual-review.json', 'test/helpers/workflow-judge-input.ts', 'test/skill-llm-eval.test.ts'],
'command reference table': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/commands.ts', 'gstack/llms.txt', 'test/skill-llm-eval.test.ts'],
'snapshot flags reference': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/snapshot.ts', 'test/skill-llm-eval.test.ts'],
'browse/SKILL.md reference': ['browse/sections/**', 'browse/SKILL.md', 'browse/SKILL.md.tmpl', 'browse/src/**', 'test/skill-llm-eval.test.ts'],
'browse/SKILL.md reference': ['browse/sections/**', 'browse/SKILL.md', 'browse/SKILL.md.tmpl', 'browse/src/**', 'SKILL.md', 'SKILL.md.tmpl', 'gstack/llms.txt', 'test/fixtures/eval-baselines.json', 'test/skill-llm-eval.test.ts'],
'setup block': ['browse/SKILL.md', 'browse/SKILL.md.tmpl', 'scripts/resolvers/aside.ts', 'scripts/resolvers/browse.ts', 'test/skill-llm-eval.test.ts'],
'regression vs baseline': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/commands.ts', 'test/fixtures/eval-baselines.json', 'test/skill-llm-eval.test.ts'],
'qa/SKILL.md workflow': ['qa/sections/**', 'qa/SKILL.md', 'qa/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'],
'qa/SKILL.md health rubric': ['qa/sections/**', 'qa/SKILL.md', 'qa/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'],
'qa/SKILL.md anti-refusal': ['qa/sections/**', 'qa/SKILL.md', 'qa/SKILL.md.tmpl', 'qa-only/SKILL.md', 'qa-only/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'],
'cross-skill greptile consistency': ['review/SKILL.md', 'review/SKILL.md.tmpl', 'ship/SKILL.md', 'ship/SKILL.md.tmpl', 'review/greptile-triage.md', 'retro/SKILL.md', 'retro/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'],
'baseline score pinning': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'test/fixtures/eval-baselines.json', 'test/skill-llm-eval.test.ts'],
// Ship & Release
'ship/SKILL.md workflow': ['ship/SKILL.md', 'ship/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts', 'test/helpers/workflow-judge-input.ts', 'test/helpers/workflow-judge-cache.ts', 'test/workflow-judge-cache.test.ts', 'scripts/eval-input-cache.ts', 'test/eval-input-cache.test.ts', 'test/workflow-judge-input.test.ts', 'test/helpers/workflow-excerpt.ts',
-1
View File
@@ -25,7 +25,6 @@ const RUNNERS = [
'test/helpers/session-runner.ts',
'test/helpers/claude-pty-runner.ts',
'test/helpers/codex-session-runner.ts',
'test/helpers/gemini-session-runner.ts',
'test/helpers/agent-sdk-runner.ts',
];
@@ -1,6 +1,7 @@
// Swift-build invariant tests. Runs against the fixture iOS app at
// test/fixtures/ios-qa/FixtureApp/. Requires the Swift toolchain
// (Xcode CLI tools or stand-alone Swift). Skipped if swift is not on PATH.
// (Xcode CLI tools or stand-alone Swift). The swift build invariants run
// only with GSTACK_TEST_SWIFT=1; the parity and harness pins always run.
//
// Two invariants:
//
@@ -298,13 +299,9 @@ describe('iOS tap harness regressions', () => {
});
});
function hasSwift(): boolean {
const r = spawnSync('swift', ['--version'], { stdio: 'pipe', timeout: 30_000 });
return r.status === 0;
}
const swiftAvailable = hasSwift();
const describeIfSwift = swiftAvailable ? describe : describe.skip;
// Explicit opt-in, not tool presence: free shards run on hosts where Swift
// may be installed, and a full swift build does not belong in every PR run.
const describeIfSwift = process.env.GSTACK_TEST_SWIFT === '1' ? describe : describe.skip;
describeIfSwift('swift build invariants', () => {
// DebugBridgeUI + DebugBridgeTouch are iOS-only (they link UIKit). Plain
@@ -1,12 +1,9 @@
// High-level E2E for /ios-qa skill flow.
//
// Two scenarios:
// 1. NO_DEVICE (gate-tier compatible): runs the gen-accessors codegen
// against a SwiftUI fixture, verifies output is correct, no daemon
// hardware required. Catches regression in source-read + codegen +
// cache + render paths without an iPhone.
// 2. WITH_DEVICE (periodic-tier, requires GSTACK_HAS_IOS_DEVICE=1): full
// daemon + tailnet + USB tunnel loop. Skipped in CI.
// Runs the gen-accessors codegen against a SwiftUI fixture and simulates the
// agent flow against the daemon with a fake device tunnel — no hardware.
// Catches regression in source-read + codegen + cache + render paths without
// an iPhone. The real-device loop lives in test/skill-e2e-ios-device.test.ts.
//
// Note: The detailed daemon HTTP unit/integration tests live next to the
// daemon source (ios-qa/daemon/test/*). This file tests the agent-flow
@@ -22,7 +19,6 @@ import type { DeviceTunnel } from '../ios-qa/daemon/src/proxy';
import { grantIdentity } from '../ios-qa/daemon/src/allowlist';
import { generate } from '../ios-qa/scripts/gen-accessors';
const HAS_DEVICE = process.env.GSTACK_HAS_IOS_DEVICE === '1';
const DEVICE_TOKEN = 'rotated-mock-bearer-token';
@@ -494,14 +490,3 @@ describe('ios-qa E2E (agent-flow simulation)', () => {
}
});
});
// ───────── WITH_DEVICE — manual smoke tests (skipped in CI) ─────────
(HAS_DEVICE ? describe : describe.skip)('ios-qa E2E (with device)', () => {
test('WITH_DEVICE: full agent loop against a real iPhone', () => {
const workDir = makeWorkDir();
// Stub — real implementation requires `devicectl` + an attached iPhone.
// Documented in ios-qa/SKILL.md.tmpl under "Manual smoke test".
expect(HAS_DEVICE).toBe(true);
});
});
File renamed without changes.
+1 -1
View File
@@ -98,7 +98,7 @@ test('the native reader cannot promote foreign cwd, child, user or tool-result t
});
test('new native annotation dependencies retain all existing observation caller owners',()=>{
const expected=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','office-hours-auto-mode','auto-decide-preserved','conductor-prose'];
const expected=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','office-hours-auto-mode','auto-decide-preserved'];
for(const file of ['test/helpers/native-auto-decide.ts','test/native-auto-decide.test.ts','test/native-auto-decide-pty.test.ts','test/fixtures/native-auto-decide-ag.json']){
const owners=Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(file)).map(([name])=>name);expect(owners).toEqual(expected);
expect(selectTests([file],E2E_TOUCHFILES).selected).toContain('auto-decide-preserved');
+2 -2
View File
@@ -27,8 +27,8 @@ test('every fixture owns one paid wrapper; public models/trials/concurrency/turn
}
expect(isPaidTestFile('test/overlay-lifecycle.test.ts')).toBe(false);
expect(OVERLAY_FIXTURES.every(f => f.trials === 10 && f.concurrency === 3)).toBe(true);
expect(OVERLAY_FIXTURES.map(f => f.maxTurns ?? 5)).toEqual([15,8,15,15,8,15]);
expect(OVERLAY_FIXTURES.map(f => f.model)).toEqual([...Array(3).fill('claude-opus-4-7'), ...Array(3).fill('claude-sonnet-4-6')]);
expect(OVERLAY_FIXTURES.map(f => f.maxTurns ?? 5)).toEqual([15,8,15,15]);
expect(OVERLAY_FIXTURES.map(f => f.model)).toEqual([...Array(3).fill('claude-opus-4-7'), 'claude-sonnet-4-6']);
expect([OVERLAY_CASE_WORK_MS, OVERLAY_RECORD_GRACE_MS, OVERLAY_CASE_OUTER_MS, OVERLAY_MIN_FILE_WALL_MS]).toEqual([1_800_000, 5_000, 1_810_000, 1_830_000]);
});
+2 -2
View File
@@ -226,11 +226,11 @@ describe('literal fixture task correctness', () => {
expect(() => assertReadOnlyWorkspace(before, snapshotWorkspace(dir))).toThrow('b');
}));
test('registry retires absent-nudge fanout cases and preserves remaining model/trial budgets', () => {
expect(OVERLAY_FIXTURES).toHaveLength(6);
expect(OVERLAY_FIXTURES).toHaveLength(4);
expect(OVERLAY_FIXTURES.some((f) => f.id.includes('fanout') || f.comparison?.unsupportedHypothesis)).toBe(false);
expect(OVERLAY_FIXTURES.every((f) => f.trials === 10 && f.comparison)).toBe(true);
expect(OVERLAY_FIXTURES.filter((f) => f.model === 'claude-opus-4-7')).toHaveLength(3);
expect(OVERLAY_FIXTURES.filter((f) => f.model === 'claude-sonnet-4-6')).toHaveLength(3);
expect(OVERLAY_FIXTURES.filter((f) => f.model === 'claude-sonnet-4-6')).toHaveLength(1);
});
});
+8 -8
View File
@@ -52,7 +52,7 @@ describe('overlay file policy', () => {
});
test('only the exact wrapper family gets one attempt and the extra process grace', () => {
expect(overlayFiles).toHaveLength(6);
expect(overlayFiles).toHaveLength(4);
expect(OVERLAY_MAX_ACTIVE_SHARDS).toBe(1);
expect(OVERLAY_MIN_FILE_WALL_MS).toBe(1_830_000);
for (const file of overlayFiles) {
@@ -123,16 +123,16 @@ describe('overlay file policy', () => {
});
describe('overlay manifest affinity and CI capacity', () => {
test('actual lifecycle wrapper guards include all six in periodic and exclude all six from gate', () => {
test('actual lifecycle wrapper guards include all four in periodic and exclude all four from gate', () => {
for (const tier of ['periodic', 'gate'] as const) {
const manifest = buildRunManifest({ tier, sliceCount: 6, evalsAll: true, env: { EVALS_ALL: '1' } });
const entries = manifest.entries.filter(entry => isOverlayTestFile(entry.file));
expect(entries).toHaveLength(6);
expect(entries).toHaveLength(4);
expect(entries.every(entry => entry.status === (tier === 'periodic' ? 'planned' : 'excluded'))).toBe(true);
}
});
test('95 files retain every case, reserve slice six, and fit 330 minutes with actual family walls', () => {
test('93 files retain every case, reserve slice six, and fit 330 minutes with actual family walls', () => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'overlay-affinity-'));
const normalFiles = Array.from({ length: 89 }, (_, i) => `test/skill-e2e-normal-${i.toString().padStart(2, '0')}.test.ts`);
const discovered = [...normalFiles, ...overlayFiles];
@@ -144,11 +144,11 @@ describe('overlay manifest affinity and CI capacity', () => {
}
const opts = { tier: 'periodic' as const, sliceCount: 6, evalsAll: true, discovered, rootDir: dir, env: { EVALS_ALL: '1' } };
const manifest = buildRunManifest(opts);
expect(manifest.entries).toHaveLength(95);
expect(new Set(manifest.entries.map(e => e.file)).size).toBe(95);
expect(manifest.entries).toHaveLength(93);
expect(new Set(manifest.entries.map(e => e.file)).size).toBe(93);
expect(manifest.entries.every(e => e.status === 'planned')).toBe(true);
const counts = [1, 2, 3, 4, 5, 6].map(slice => manifest.entries.filter(e => e.slice === slice).length);
expect(counts).toEqual([18, 18, 18, 18, 17, 6]);
expect(counts).toEqual([18, 18, 18, 18, 17, 4]);
expect(manifest.entries.filter(e => e.slice === 6).map(e => e.file).sort()).toEqual([...overlayFiles].sort());
expect(buildRunManifest({ ...opts, discovered: [...discovered].reverse() })).toEqual(manifest);
expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest);
@@ -172,7 +172,7 @@ describe('overlay manifest affinity and CI capacity', () => {
const overlayMinutes = Math.ceil(overlayFiles.length / OVERLAY_MAX_ACTIVE_SHARDS)
* Math.max(...overlayFiles.map(file => resolvePaidShardTimeoutMs([file]))) / 60_000;
expect(normalMinutes).toBe(270);
expect(overlayMinutes).toBe(183);
expect(overlayMinutes).toBe(122);
expect(job['timeout-minutes']).toBeGreaterThanOrEqual(Math.max(normalMinutes, overlayMinutes) + 20);
expect(job['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(AUTOPLAN_CHAIN_BUDGET.ciJobMs);
+5 -5
View File
@@ -18,7 +18,7 @@ import { resolveModuleSelection } from './helpers/e2e-helpers';
const ROOT = path.resolve(import.meta.dir, '..');
const CEO_FILES = [
'test/skill-e2e-plan.test.ts', 'test/skill-e2e-ask-user-question-format-compliance.test.ts',
'test/skill-e2e-opus-47.test.ts', 'test/skill-llm-eval.test.ts',
'test/skill-e2e-retro.test.ts', 'test/skill-llm-eval.test.ts',
];
const ceoManifest = () => buildRunManifest({ tier: 'gate', profile: 'pr', sliceCount: 1,
evalsAll: false, env: {}, changedFiles: ['plan-ceo-review/SKILL.md.tmpl'], discovered: CEO_FILES });
@@ -70,7 +70,7 @@ describe('PR profile paid-runner integration', () => {
expect(manifest.selection?.e2e).toContain('plan-ceo-review-benefits');
expect(manifest.selection?.e2e).not.toContain('plan-ceo-review-plan-mode');
expect(manifest.selection?.judges).toContain('plan-ceo-review/SKILL.md modes');
expect(manifest.entries.find(entry => entry.file.includes('opus-47'))?.status).toBe('skipped-by-diff');
expect(manifest.entries.find(entry => entry.file.includes('skill-e2e-retro'))?.status).toBe('skipped-by-diff');
expect(manifest.entries.find(entry => entry.file.includes('ask-user-question'))?.status).toBe('planned');
expect(manifest.prCoverage?.deferred.some(item => item.id === 'plan-ceo-review-plan-mode')).toBe(true);
expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest);
@@ -145,8 +145,8 @@ describe('PR profile paid-runner integration', () => {
broad.selection!.e2e!.push('qa-only-no-fix'); broad.prCoverage!.e2e.push('qa-only-no-fix');
expect(() => parseRunManifest(JSON.stringify(broad))).toThrow('broad-only');
const injected = structuredClone(manifest);
injected.entries.find(entry => entry.file.includes('opus-47'))!.status = 'planned';
injected.entries.find(entry => entry.file.includes('opus-47'))!.slice = 1;
injected.entries.find(entry => entry.file.includes('skill-e2e-retro'))!.status = 'planned';
injected.entries.find(entry => entry.file.includes('skill-e2e-retro'))!.slice = 1;
expect(() => parseRunManifest(JSON.stringify(injected))).toThrow('outside its PR case selection');
for (const action of ['remove', 'skip', 'duplicate'] as const) {
const missing = structuredClone(manifest);
@@ -184,7 +184,7 @@ describe('PR profile paid-runner integration', () => {
const judge = new RegExp(prProfileTestNamePattern('test/skill-llm-eval.test.ts', selection));
expect(judge.test('LLM-as-judge plan-ceo-review/SKILL.md modes')).toBe(true);
expect(judge.test('LLM-as-judge plan-ceo-review/SKILLxmd modes')).toBe(false);
expect(prProfileFileSelected('test/skill-e2e-opus-47.test.ts', selection)).toBe(false);
expect(prProfileFileSelected('test/skill-e2e-retro.test.ts', selection)).toBe(false);
});
test('real Bun child executes only the persisted case through the actual registered helper', async () => {
+10 -13
View File
@@ -13,9 +13,8 @@ import {
const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8');
const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file));
const expectedWalls = {
'test/skill-llm-eval.test.ts': 6_920_000,
'test/skill-llm-eval.test.ts': 5_960_000,
'test/skill-e2e-auq-consistency.test.ts': 2_040_000,
'test/codex-e2e-plan-format.test.ts': 5_000_000,
'test/skill-e2e-auq-matrix.test.ts': 3_720_000,
'test/skill-e2e-plan-format.test.ts': 2_600_000,
'test/skill-e2e-auto-decide-preserved.test.ts': 1_920_000,
@@ -30,9 +29,9 @@ const expectedWalls = {
'test/skill-e2e-plan.test.ts': 7_320_000,
};
test('registration covers exactly the fifteen demonstrated full-file retry gaps', () => {
test('registration covers exactly the fourteen demonstrated full-file retry gaps', () => {
expect(Object.fromEntries(newBudgets.map(row => [row.file, row.shardMs]))).toEqual(expectedWalls);
expect(new Set(FILE_RETRY_BUDGETS.map(row => row.file)).size).toBe(21);
expect(new Set(FILE_RETRY_BUDGETS.map(row => row.file)).size).toBe(20);
expect(STRICT_RETRY_CASE_BUDGETS.map(row => row.file)).toEqual([
...FINDING_RETRY_BUDGETS.map(row => row.file), AUQ_CONSISTENCY_RETRY_BUDGET.file,
]);
@@ -51,8 +50,6 @@ test('source allowances retain all captures, cases, and finalization grace', ()
expect(auq).toContain('for (const [i, capture] of captures.entries())');
expect(auq).toContain('N_RUNS * CAPTURE_MS + 60_000');
expect(AUQ_CONSISTENCY_RETRY_BUDGET.testMs).toBe(960_000);
expect(timeoutExpressions('test/codex-e2e-plan-format.test.ts')).toEqual(Array(4).fill('CAPTURE_LONG_MS'));
expect(read('test/codex-e2e-plan-format.test.ts')).toContain('test(name, fn, timeout + CODEX_EVAL_FINALIZE_MS)');
expect(read('test/helpers/codex-eval.ts')).toContain('CODEX_EVAL_FINALIZE_MS = 2 * CODEX_DRAIN_GRACE_MS');
expect(read('test/helpers/codex-session-runner.ts')).toMatch(/CODEX_DRAIN_GRACE_MS\s*=\s*5_000/);
expect(read('test/helpers/office-hours-attempt.ts')).toContain('OFFICE_HOURS_BUN_GRACE_MS = 10_000');
@@ -140,15 +137,15 @@ test('quality judge supervision includes the added judge without changing ordina
expect(ALL_TIERS).toEqual({ JUDGE_MS: 120000, CAPTURE_MS: 300000, CAPTURE_LONG_MS: 600000, PTY_MS: 900000, PTY_LONG_MS: 1200000 });
const quality = 'test/skill-llm-eval.test.ts';
const qualityBudget = FILE_RETRY_BUDGETS.find(row => row.file === quality)!;
expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 6_920_000, source: 'registered', policyId: qualityBudget.id });
expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 5_960_000, source: 'registered', policyId: qualityBudget.id });
expect(retriesForFiles([quality])).toBe(1);
const qualitySource = read(quality);
const judgeTimeouts = [...qualitySource.matchAll(/}\s*,\s*(JUDGE_MS|WORKFLOW_JUDGE_TEST_MS)\s*\);/g)].map(match => match[1]);
expect(judgeTimeouts.filter(timeout => timeout === 'JUDGE_MS')).toHaveLength(11);
expect(judgeTimeouts.filter(timeout => timeout === 'JUDGE_MS')).toHaveLength(7);
expect(judgeTimeouts.filter(timeout => timeout === 'WORKFLOW_JUDGE_TEST_MS')).toHaveLength(16);
expect(qualitySource).toContain('WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000');
expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS');
expect(qualityBudget.shardMs).toBe((11 * ALL_TIERS.JUDGE_MS + 16 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000);
expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 16 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000);
expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([
[2, 1500000, 1, 6120000], ...Array(5).fill([1, 1500000, 1, 3120000]),
]);
@@ -213,9 +210,9 @@ test('both gate executors cover the complete census without increasing aggregate
expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: slices }, (_, i) => i + 1));
expect(planned.slices).toBe(slices);
const manifest = buildRunManifest({ tier: 'gate', sliceCount: planned.slices, evalsAll: true, env: { EVALS_ALL: '1' } });
expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(58);
expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(52);
const files = manifest.entries.filter(row => row.status === 'planned').map(row => row.file);
expect(new Set(files).size).toBe(58);
expect(new Set(files).size).toBe(52);
expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected.sort());
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
manifest.entries.filter(row => row.status === 'planned' && row.slice === slice).map(row => row.file), workers,
@@ -250,8 +247,8 @@ test('the periodic executor supervises every actual case and retry within its CI
const manifest = buildRunManifest({ tier: 'periodic', sliceCount: planned.slices,
dedicatedAutoplanSlice: planned.dedicatedAutoplanSlice, evalsAll: true, env: { EVALS_ALL: '1' } });
const census = manifest.entries.filter(row => row.status === 'planned');
expect(census).toHaveLength(100);
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_920_000);
expect(census).toHaveLength(82);
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(5_960_000);
expect(manifest.autoplanSlice).toBe(8);
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
census.filter(row => row.slice === slice).map(row => row.file), active.jobs,
+1 -1
View File
@@ -60,7 +60,7 @@ describe('paid test enumeration', () => {
const files = collectPaidTestFiles();
expect(files.length).toBeGreaterThan(0);
expect(files.every(isPaidTestFile)).toBe(true);
expect(PAID_TEST_GLOBS.length).toBe(7);
expect(PAID_TEST_GLOBS.length).toBe(6);
const shards = planPaidShards(files);
expect(shards.flat().sort()).toEqual([...files].sort());
+15 -26
View File
@@ -93,7 +93,7 @@ describe('periodic fixture dependencies select their behavioral cases', () => {
['test/fixtures/ceo-current-record-6aef.json', ['plan-ceo-finding-count']],
['test/fixtures/ceo-payment-ledger-decisions.json', ['plan-ceo-finding-count']],
['test/setup-gbrain-remote-caller.test.ts', ['setup-gbrain-remote']],
['test/skill-fixture.test.ts', ['journey-ideation', 'journey-plan-eng', 'journey-debug', 'journey-qa', 'journey-code-review', 'journey-ship', 'journey-docs', 'journey-retro', 'journey-design-system', 'journey-visual-qa']],
['test/skill-fixture.test.ts', ['journey-ideation', 'journey-plan-eng', 'journey-debug', 'journey-qa', 'journey-code-review', 'journey-ship', 'journey-docs', 'journey-retro', 'journey-design-system', 'journey-visual-qa', 'journey-negatives']],
['test/office-hours-writeback-env.test.ts', ['office-hours-brain-writeback']],
['test/helpers/setup-gbrain-sandbox.ts', ['setup-gbrain-bad-token', 'setup-gbrain-path4-local-pglite', 'setup-gbrain-remote']],
['test/helpers/setup-gbrain-fixture-command.ts', ['setup-gbrain-bad-token', 'setup-gbrain-path4-local-pglite']],
@@ -154,7 +154,7 @@ describe('periodic fixture dependencies select their behavioral cases', () => {
}
test('SDK runner changes retain the native gate and existing periodic consumers', () => {
const periodic = ['brain-privacy-gate', 'setup-gbrain-remote', 'setup-gbrain-bad-token',
const periodic = ['setup-gbrain-remote', 'setup-gbrain-bad-token',
'setup-gbrain-path4-local-pglite', ...OVERLAY_FIXTURES.map(fixture => `overlay-harness-${fixture.id}`)];
const result = selectTests(['test/agent-sdk-runner.test.ts'], E2E_TOUCHFILES);
expect(result.reason).toBe('diff');
@@ -251,9 +251,9 @@ test('native fixture dependencies include the migrated auto-decision and seeded
test('shared native input dependencies select every PTY consumer without changing tiers', () => {
const expected = selectTests(['test/helpers/claude-pty-runner.ts'], E2E_TOUCHFILES).selected.sort();
expect(expected).toHaveLength(22);
expect(expected).toHaveLength(20);
expect(expected.filter(id => E2E_TIERS[id] === 'gate')).toHaveLength(7);
expect(expected.filter(id => E2E_TIERS[id] === 'periodic')).toHaveLength(15);
expect(expected.filter(id => E2E_TIERS[id] === 'periodic')).toHaveLength(13);
for (const file of ['test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts',
'test/helpers/plan-skill-questions.ts', 'test/plan-skill-questions.test.ts', 'test/fixtures/design-tasks-bash-permission.json', 'test/fixtures/eng-auq-validation-error.json',
'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts',
@@ -266,14 +266,14 @@ test('shared native input dependencies select every PTY consumer without changin
test('seed submission dependencies select every seeded caller with its existing tier', () => {
const expected = ['auto-decide-preserved', 'conductor-prose', 'plan-ceo-review-plan-mode',
const expected = ['auto-decide-preserved', 'plan-ceo-review-plan-mode',
'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-mode-no-op'];
for (const file of ['test/helpers/fake-plan-seed.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts']) {
const result = selectTests([file], E2E_TOUCHFILES);
expect(result.reason).toBe('diff');
expect(result.selected.sort()).toEqual(expected);
}
expect(expected.map(id => E2E_TIERS[id])).toEqual(['periodic', 'periodic', 'gate', 'periodic', 'gate', 'periodic', 'gate']);
expect(expected.map(id => E2E_TIERS[id])).toEqual(['periodic', 'gate', 'periodic', 'gate', 'periodic', 'gate']);
});
test('task emission source selects CEO completion consumers', () => {
@@ -323,7 +323,6 @@ test('Eng approval-rule source and free contract controls select every declared
'plan-review-report',
'plan-eng-review-plan-mode',
'plan-mode-no-op',
'conductor-prose',
'carve-section-loading',
'autoplan-chain-pty',
'plan-eng-finding-count',
@@ -334,8 +333,6 @@ test('Eng approval-rule source and free contract controls select every declared
'plan-ceo-review-prosons-cadence',
'plan-review-prosons-format',
'codex-offered-eng-review',
'codex-plan-eng-format-coverage',
'codex-plan-eng-format-kind',
'plan-eng-coverage-audit',
'autoplan-dual-voice'
];
@@ -363,8 +360,7 @@ test('native compact-boundary ancestry selects every consuming callback', () =>
'plan-eng-finding-count', 'plan-design-finding-count', 'plan-devex-finding-count',
'plan-eng-multi-finding-batching', 'plan-ceo-split-overflow',
'plan-design-with-ui-scope', 'plan-design-review-plan-mode', 'plan-eng-review-plan-mode',
'auto-decide-preserved', 'conductor-prose',
].sort();
'auto-decide-preserved', ].sort();
for (const file of ['test/helpers/plan-count-transcript.ts', 'test/plan-count-session-cwd.test.ts']) {
const selected = selectTests([file], E2E_TOUCHFILES);
expect(selected.reason).toBe('diff');
@@ -383,7 +379,7 @@ test('same-plan expansion disposition replay selects the existing mode helper co
test('structured auto-decision evidence selects every native observer', () => {
const expected = ['auto-decide-preserved', 'conductor-prose', 'plan-ceo-review-plan-mode',
const expected = ['auto-decide-preserved', 'plan-ceo-review-plan-mode',
'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-mode-no-op'];
for (const file of ['test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json',
'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json']) {
@@ -398,7 +394,7 @@ test('structured auto-decision evidence selects every native observer', () => {
test('explanatory native mode evidence selects all observers with their existing tiers', () => {
const expected = ['auto-decide-preserved', 'conductor-prose', 'office-hours-auto-mode',
const expected = ['auto-decide-preserved', 'office-hours-auto-mode',
'plan-ceo-review-plan-mode', 'plan-design-review-plan-mode', 'plan-devex-review-plan-mode',
'plan-eng-review-plan-mode', 'plan-mode-no-op'];
for (const file of ['test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts',
@@ -408,7 +404,7 @@ test('explanatory native mode evidence selects all observers with their existing
expect(selectTests([file], LLM_JUDGE_TOUCHFILES).selected).toEqual([]);
}
expect(expected.map(id => E2E_TIERS[id])).toEqual([
'periodic', 'periodic', 'gate', 'gate', 'periodic', 'gate', 'periodic', 'gate',
'periodic', 'gate', 'gate', 'periodic', 'gate', 'periodic', 'gate',
]);
});
@@ -432,8 +428,7 @@ test('file supervision regression selects all affected callers with their existi
'plan-ceo-review-benefits', 'plan-devex-finding-floor', 'plan-mode-no-op', 'plan-review-report',
];
const periodic = [
'auto-decide-preserved', 'codex-plan-ceo-format-approach', 'codex-plan-ceo-format-mode',
'codex-plan-eng-format-coverage', 'codex-plan-eng-format-kind', 'plan-ceo-mode-routing',
'auto-decide-preserved', 'plan-ceo-mode-routing',
'plan-ceo-review', 'plan-ceo-review-expansion-energy', 'plan-ceo-review-format-approach',
'plan-ceo-review-format-mode', 'plan-ceo-review-prosons-cadence', 'plan-ceo-review-selective',
'plan-design-finding-floor', 'plan-eng-finding-floor', 'plan-eng-review', 'plan-eng-review-artifact',
@@ -498,7 +493,6 @@ const nativeRepairDependencies = [
"plan-mode-no-op",
"office-hours-auto-mode",
"auto-decide-preserved",
"conductor-prose"
]
},
{
@@ -595,10 +589,8 @@ const nativeRepairDependencies = [
"plan-mode-no-op",
"office-hours-auto-mode",
"auto-decide-preserved",
"conductor-prose",
"plan-ceo-mode-routing",
"plan-design-with-ui-scope",
"ship-idempotency-pty",
"autoplan-chain-pty",
"plan-ceo-finding-count",
"plan-eng-finding-count",
@@ -632,7 +624,6 @@ test('native repair dependencies preserve every original tier', () => {
"plan-mode-no-op": "gate",
"office-hours-auto-mode": "gate",
"auto-decide-preserved": "periodic",
"conductor-prose": "periodic",
"plan-ceo-finding-count": "periodic",
"plan-eng-finding-count": "periodic",
"plan-design-finding-count": "periodic",
@@ -657,7 +648,6 @@ test('promoted public transcript decoder keeps its actual callers selected', ()
'plan-eng-review-plan-mode',
'plan-design-review-plan-mode',
'auto-decide-preserved',
'conductor-prose',
'plan-ceo-mode-routing',
'plan-design-with-ui-scope',
'autoplan-chain-pty',
@@ -749,7 +739,7 @@ test('numbered native-menu captures select the existing parser consumers', () =>
const expected = Object.entries(E2E_TOUCHFILES)
.filter(([, files]) => files.includes('test/plan-skill-questions.test.ts'))
.map(([id]) => id).sort();
expect(expected).toHaveLength(22);
expect(expected).toHaveLength(20);
for (const file of ['test/pty-numbered-option-indent-native.test.ts',
'test/fixtures/ceo-split-e5-numbered-description-491.json']) {
expect(selectTests([file], E2E_TOUCHFILES).selected.sort()).toEqual(expected);
@@ -808,7 +798,7 @@ test('stderr lifecycle regression selects runtime consumers without a quality-ma
'benchmark-workflow', 'setup-deploy-workflow', 'autoplan-dual-voice', 'scrape-match-path', 'scrape-prototype-path',
'skillify-happy-path', 'skillify-provenance-refusal', 'skillify-approval-reject', 'journey-ideation', 'journey-plan-eng',
'journey-debug', 'journey-qa', 'journey-code-review', 'journey-ship', 'journey-docs',
'journey-retro', 'journey-design-system', 'journey-visual-qa', 'fanout-arm-overlay-on', 'fanout-arm-overlay-off',
'journey-retro', 'journey-design-system', 'journey-visual-qa', 'journey-negatives',
'office-hours-brain-writeback', 'arm-benchmark-native-overbuild', 'arm-benchmark-crud-endpoint', 'arm-benchmark-bugfix-decoys', 'office-hours-section-loading',
];
const file = 'test/session-runner-stream-lifecycle.test.ts';
@@ -831,8 +821,7 @@ for (const file of ['test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures
const selected = selectTests([file], E2E_TOUCHFILES);
expect(selected.reason).toBe('diff');
expect(selected.selected.sort()).toEqual([
'auto-decide-preserved', 'autoplan-chain-pty', 'conductor-prose',
'plan-ceo-finding-count', 'plan-ceo-mode-routing', 'plan-ceo-split-overflow',
'auto-decide-preserved', 'autoplan-chain-pty', 'plan-ceo-finding-count', 'plan-ceo-mode-routing', 'plan-ceo-split-overflow',
'plan-design-finding-count', 'plan-design-review-plan-mode', 'plan-design-with-ui-scope',
'plan-devex-finding-count', 'plan-eng-finding-count', 'plan-eng-multi-finding-batching',
'plan-eng-review-plan-mode',
@@ -844,7 +833,7 @@ for (const file of ['test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures
test('native clipped regressions retain the existing parser and owned-permission selection', () => {
for (const [dependency, count, files] of [
['test/helpers/claude-pty-runner.ts', 22, [
['test/helpers/claude-pty-runner.ts', 20, [
'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json',
]],
['test/helpers/plan-count-file-permission.ts', 10, [
+3 -3
View File
@@ -11,9 +11,9 @@ const ROOT = path.resolve(import.meta.dir, '..');
const source = fs.readFileSync(path.join(ROOT, 'test/skill-e2e-design.test.ts'), 'utf8');
const id = 'plan-design-review-plan-mode';
// Same source-evaluation pattern as plan-tune-cathedral-fixture.test.ts: run
// the real suite registration and selected callback, without importing paid
// initialization. All filesystem mutations stay in this standalone fixture.
// Source-evaluation pattern: run the real suite registration and selected
// callback, without importing paid initialization. All filesystem mutations
// stay in this standalone fixture.
async function exercise(mode: 'success' | 'max-turns' | 'first-timeout' | 'second-timeout' | 'saved-timeout' | 'empty-summary' | 'write-failure' | 'unchanged-seed' | 'no-additions' | 'short-plan' | 'api-error' | 'plan-read-failure' | 'attempt-deadline') {
const scratch = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'legacy-design-free-')));
const home = path.join(scratch, 'home'); fs.mkdirSync(home);
+2 -3
View File
@@ -26,12 +26,11 @@ test('semantic helper changes also select the separate DX analysis calibration',
// Its direct behavioral consumers extend the unchanged helper-only set.
if (file === 'test/plan-review-cases.test.ts') expected.push(
'plan-eng-review', 'plan-eng-review-artifact', 'plan-review-report',
'plan-eng-review-plan-mode', 'plan-mode-no-op', 'conductor-prose',
'plan-eng-review-plan-mode', 'plan-mode-no-op',
'carve-section-loading', 'autoplan-chain-pty', 'plan-eng-finding-floor',
'plan-eng-review-format-coverage', 'plan-eng-review-format-kind',
'plan-ceo-review-prosons-cadence', 'plan-review-prosons-format',
'codex-offered-eng-review', 'codex-plan-eng-format-coverage',
'codex-plan-eng-format-kind', 'plan-eng-coverage-audit', 'autoplan-dual-voice',
'codex-offered-eng-review', 'plan-eng-coverage-audit', 'autoplan-dual-voice',
);
expect(selectTests([file], E2E_TOUCHFILES, []).selected.sort()).toEqual(
expected.sort());
-129
View File
@@ -1,129 +0,0 @@
import { expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as path from 'node:path';
import * as os from 'node:os';
import { spawnSync } from 'node:child_process';
import { E2E_TOUCHFILES } from './helpers/touchfiles';
const ROOT = path.resolve(import.meta.dir, '..');
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-e2e-plan-tune-cathedral.test.ts'), 'utf8');
const names = ['plan-tune-hook-capture', 'plan-tune-enforcement', 'plan-tune-annotation', 'plan-tune-codex-import', 'plan-tune-dream-cycle'];
async function exercise(selected = names, fault?: 'missing-log-lib' | 'missing-hook-lib' | 'first-hook' | 'setup', hostEnv: Record<string, string> = {}) {
// Execute the actual selected callbacks with real local bins. Never import
// E2E initialization, call a provider, or pass ambient auth/host state.
const scratch = fs.mkdtempSync(path.join(os.tmpdir(), 'cathedral-contract-'));
const suites: any[] = [], finalizers: any[] = [], rows: any[] = [], attempts: any[] = [], dirs: string[] = [], removals: any[] = [];
const invoked: string[] = [];
let current: any, failedHook = false;
const owned = (p: string) => { if (!p.startsWith(scratch + path.sep)) throw new Error('Foreign fixture path'); };
const env = { PATH: process.env.PATH, HOME: path.join(scratch, 'home'), GIT_CONFIG_NOSYSTEM: '1', GIT_CONFIG_GLOBAL: process.platform === 'win32' ? 'NUL' : '/dev/null', ...hostEnv };
fs.mkdirSync(env.HOME);
const args: Record<string, any> = {
ROOT, path, os: {tmpdir:()=>scratch}, process: {env}, expect,
beforeAll: (fn: any) => current.setups.push(fn),
afterAll: (fn: any) => current ? current.teardowns.push(fn) : finalizers.push(fn),
describeIfSelected: (_title: string, keys: string[], fn: any) => {
if (!keys.some(key => selected.includes(key))) return;
const suite = {setups:[],teardowns:[],tests:[]}; suites.push(suite); current=suite; fn(); current=undefined;
},
testConcurrentIfSelected: (name: string, fn: any) => { if(selected.includes(name)) current.tests.push({name,fn}); },
createEvalCollector: () => ({addTest:(row:any)=>rows.push(row)}), finalizeEvalCollector:()=>{},
copyDirSync: (a: string, b: string) => { owned(b); fs.cpSync(a,b,{recursive:true}); },
fs: { ...fs,
mkdtempSync(prefix: string) { owned(prefix); const dir=fs.mkdtempSync(prefix); dirs.push(dir); return dir; },
copyFileSync(a: string,b: string) {
owned(b);
if(fault==='setup') throw new Error('Synthetic fixture copy failure');
if((fault==='missing-log-lib' && a.endsWith('/lib/jsonl-store.ts')) || (fault==='missing-hook-lib' && a.endsWith('/lib/is-conductor.ts'))) return;
fs.copyFileSync(a,b);
},
rmSync(p: string,opts: any) {
owned(p); const memory=path.join(p,'.gstack-state/free-text-memory.json');
removals.push({path:p,nuggets:fs.existsSync(memory)?JSON.parse(fs.readFileSync(memory,'utf8')).nuggets.length:null});
fs.rmSync(p,opts);
},
},
spawnSync: (bin: string,argv: string[],opts: any) => {
if(bin!=='git') {
owned(bin);
expect(['question-log-hook','question-preference-hook','gstack-codex-session-import','gstack-distill-apply']).toContain(path.basename(bin));
owned(opts.env.GSTACK_STATE_ROOT);
}
if(opts.cwd) owned(opts.cwd);
expect(Number.isFinite(opts.timeout) && opts.timeout > 0).toBe(true);
invoked.push(path.basename(bin));
if(fault==='first-hook' && path.basename(bin)==='question-preference-hook' && !failedHook) {
failedHook=true; return {status:1,stdout:'',stderr:'Synthetic transient hook failure'};
}
return spawnSync(bin,argv,{...opts,timeout:opts.timeout,env:opts.env??env});
},
};
let body=source; for(const m of source.matchAll(/^import[\s\S]*?;\n/gm)) body=body.replace(m[0],'');
try {
new Function(...Object.keys(args),new Bun.Transpiler({loader:'ts'}).transformSync(body))(...Object.values(args));
for(const suite of suites) {
for(const setup of suite.setups) await setup();
for(const callback of suite.tests) for(let attempt=1;attempt<=2;attempt++) {
let error: unknown; try {await callback.fn();} catch(e) {error=e;}
attempts.push({name:callback.name,attempt,error,remaining:dirs.filter(p=>fs.existsSync(p))});
}
for(const done of suite.teardowns) await done();
}
for(const done of finalizers) await done();
return {rows,attempts,dirs,removals,invoked,allRemoved:dirs.every(p=>!fs.existsSync(p))};
} finally {fs.rmSync(scratch,{recursive:true,force:true});}
}
test('Cathedral callbacks run all five real local contracts with fresh attempts and truthful rows', async () => {
const x=await exercise();
expect(x.attempts).toHaveLength(10); expect(x.rows).toHaveLength(10); expect(x.dirs).toHaveLength(10);
expect(new Set(x.dirs).size).toBe(10); expect(x.allRemoved).toBe(true);
for(const attempt of x.attempts) {expect(attempt.error).toBeUndefined();expect(attempt.remaining).toEqual([]);}
for(const row of x.rows) {expect(row.passed).toBe(true);expect(row.cost_usd).toBe(0);expect(row.output).toContain('no model invocation');}
});
test('Missing installed libraries still fail the actual hook and log contracts', async () => {
for(const [name,fault] of [['plan-tune-hook-capture','missing-log-lib'],['plan-tune-enforcement','missing-hook-lib']] as const) {
const x=await exercise([name],fault);
expect(x.rows).toHaveLength(2); expect(x.allRemoved).toBe(true);
for(const attempt of x.attempts) expect(attempt.error).toBeDefined();
expect(x.rows.map(row=>row.passed)).toEqual([false,false]);
}
});
test('A failed dream hook retries with one fresh nugget and preserves the failed attempt', async () => {
const x=await exercise(['plan-tune-dream-cycle'],'first-hook');
expect(x.attempts[0].error).toBeDefined(); expect(x.attempts[1].error).toBeUndefined();
expect(x.rows.map(row=>row.passed)).toEqual([false,true]);
expect(x.removals.map(row=>row.nuggets)).toEqual([1,1]);
expect(new Set(x.dirs).size).toBe(2); expect(x.allRemoved).toBe(true);
});
test('A partial fixture setup failure is recorded once and cleans only its owned directory', async () => {
const x=await exercise(['plan-tune-hook-capture'],'setup');
expect(x.rows.map(row=>row.passed)).toEqual([false,false]); expect(x.dirs).toHaveLength(2); expect(x.removals).toHaveLength(2);
expect(x.invoked.every(bin=>bin==='git')).toBe(true); expect(x.allRemoved).toBe(true);
});
test('Cathedral fixture controls and copied libraries select all five existing owners only', () => {
for(const file of ['test/plan-tune-cathedral-fixture.test.ts','lib/jsonl-store.ts','lib/is-conductor.ts']) {
const owners=Object.entries(E2E_TOUCHFILES).filter(([name,paths])=>names.includes(name)&&paths.includes(file)).map(([name])=>name);
expect(owners).toEqual(names);
}
expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes('test/plan-tune-cathedral-fixture.test.ts')).map(([name])=>name)).toEqual(names);
});
test('Plain Claude cathedral contracts stay isolated from either inherited Conductor marker', async () => {
for (const hostEnv of [
{ CONDUCTOR_WORKSPACE_PATH: '/synthetic/conductor/workspace' },
{ CONDUCTOR_PORT: '55070', GSTACK_SESSION_KIND: 'spawned', OPENCLAW_SESSION: 'synthetic-session' },
]) {
const x = await exercise(['plan-tune-annotation', 'plan-tune-dream-cycle'], undefined, hostEnv);
expect(x.attempts).toHaveLength(4);
expect(x.rows.map(row => row.passed)).toEqual([true, true, true, true]);
for (const attempt of x.attempts) expect(attempt.error).toBeUndefined();
expect(x.allRemoved).toBe(true);
}
});
@@ -1,42 +1,22 @@
/**
* /plan-tune cathedral E2E (T16) — 5 scenarios, all gate tier per D12.
* /plan-tune cathedral contract tests (T16) — 5 scenarios, free suite.
*
* Each scenario verifies that the cathedral's substrate works end-to-end
* through local hook and bin invocations. No model is called: these scenarios
* exercise the installed-file contracts, using synthetic hook envelopes and
* a synthetic Codex session. Unit tests cover the individual components.
*
* Touchfile registration in test/helpers/touchfiles.ts:
* - plan-tune-hook-capture
* - plan-tune-enforcement
* - plan-tune-annotation
* - plan-tune-codex-import
* - plan-tune-dream-cycle
*
* Each scenario uses GSTACK_STATE_ROOT to isolate from the user's real
* ~/.gstack (per cathedral T1 + Codex D16 fix). Every attempt gets a fresh fixture.
*/
import { afterAll, expect } from 'bun:test';
import {
ROOT,
describeIfSelected,
testConcurrentIfSelected,
copyDirSync,
createEvalCollector,
finalizeEvalCollector,
} from './helpers/e2e-helpers';
import { describe, expect, test } from 'bun:test';
import { ROOT, copyDirSync } from './helpers/e2e-helpers';
import { spawnSync } from 'child_process';
import * as fs from 'fs';
import * as path from 'path';
import * as os from 'os';
const collector = createEvalCollector('e2e-plan-tune-cathedral');
afterAll(() => {
finalizeEvalCollector(collector);
});
/** Scaffold a fixture project with the bins + scripts the cathedral needs. */
function scaffoldFixture(workDir: string): { workDir: string; stateRoot: string; slug: string; env: NodeJS.ProcessEnv } {
const stateRoot = path.join(workDir, '.gstack-state');
@@ -114,27 +94,18 @@ function cleanupFixture(workDir: string): void {
}
}
/** Own setup, assertions and cleanup inside each callback, including Bun retries. */
/** Own setup, assertions and cleanup inside each callback. */
function testCathedral(
name: string,
prefix: string,
run: (fixture: ReturnType<typeof scaffoldFixture>) => Promise<void>,
): void {
testConcurrentIfSelected(name, async () => {
const started = Date.now();
let workDir: string | undefined;
let passed = false;
test(name, async () => {
const workDir = fs.mkdtempSync(path.join(os.tmpdir(), prefix));
try {
workDir = fs.mkdtempSync(path.join(os.tmpdir(), prefix));
await run(scaffoldFixture(workDir));
passed = true;
} finally {
if (workDir) cleanupFixture(workDir);
collector?.addTest({
name, suite: 'plan-tune-cathedral', tier: 'e2e', passed,
duration_ms: Date.now() - started, cost_usd: 0,
output: 'Local hook/bin contract; no model invocation.',
});
cleanupFixture(workDir);
}
});
}
@@ -143,7 +114,7 @@ function testCathedral(
// Scenario 1: Hook capture — PostToolUse hook writes to question-log.jsonl
// ---------------------------------------------------------------------------
describeIfSelected('PlanTune cathedral E2E: hook capture', ['plan-tune-hook-capture'], () => {
describe('PlanTune cathedral E2E: hook capture', () => {
testCathedral('plan-tune-hook-capture', 'cathedral-cap-', async (fixture) => {
// Direct hook invocation simulates Claude Code's PostToolUse delivery.
// E2E verifies the hook + bin chain works against real bins on disk
@@ -190,7 +161,7 @@ describeIfSelected('PlanTune cathedral E2E: hook capture', ['plan-tune-hook-capt
// Scenario 2: Enforcement — never-ask preference + marker + 2-way → deny
// ---------------------------------------------------------------------------
describeIfSelected('PlanTune cathedral E2E: enforcement', ['plan-tune-enforcement'], () => {
describe('PlanTune cathedral E2E: enforcement', () => {
testCathedral('plan-tune-enforcement', 'cathedral-enf-', async (fixture) => {
fs.mkdirSync(path.join(fixture.stateRoot, 'projects', fixture.slug), { recursive: true });
fs.writeFileSync(
@@ -253,7 +224,7 @@ describeIfSelected('PlanTune cathedral E2E: enforcement', ['plan-tune-enforcemen
// Scenario 3: Annotation — declared profile injected via additionalContext
// ---------------------------------------------------------------------------
describeIfSelected('PlanTune cathedral E2E: annotation', ['plan-tune-annotation'], () => {
describe('PlanTune cathedral E2E: annotation', () => {
testCathedral('plan-tune-annotation', 'cathedral-ann-', async (fixture) => {
// Strong declared profile that should annotate any signal_key=detail-preference question.
fs.writeFileSync(
@@ -318,7 +289,7 @@ describeIfSelected('PlanTune cathedral E2E: annotation', ['plan-tune-annotation'
// Scenario 4: Codex import — JSONL session → import bin → log fills
// ---------------------------------------------------------------------------
describeIfSelected('PlanTune cathedral E2E: codex import', ['plan-tune-codex-import'], () => {
describe('PlanTune cathedral E2E: codex import', () => {
testCathedral('plan-tune-codex-import', 'cathedral-cdx-', async (fixture) => {
const sessionFile = path.join(fixture.workDir, 'rollout-cathedral.jsonl');
const lines = [
@@ -374,7 +345,7 @@ describeIfSelected('PlanTune cathedral E2E: codex import', ['plan-tune-codex-imp
// re-fire → memory injection
// ---------------------------------------------------------------------------
describeIfSelected('PlanTune cathedral E2E: dream cycle', ['plan-tune-dream-cycle'], () => {
describe('PlanTune cathedral E2E: dream cycle', () => {
testCathedral('plan-tune-dream-cycle', 'cathedral-dream-', async (fixture) => {
// Seed proposals file directly (the SDK call is exercised by the unit
// test; here we verify apply → re-fire round-trip on top of a known
+2 -3
View File
@@ -87,14 +87,13 @@ describe('session-runner timeout kills the whole process group', () => {
}, 60_000);
});
describe('all three provider runners carry the group-kill wiring', () => {
// Source pin, not behavior: codex/gemini need their real binaries for a
describe('both CLI provider runners carry the group-kill wiring', () => {
// Source pin, not behavior: codex needs its real binary for a
// behavioral run, but the kill wiring is identical code — a runner that
// drops `detached` or reverts to a bare kill() re-opens the orphan class.
const runners = [
'test/helpers/session-runner.ts',
'test/helpers/codex-session-runner.ts',
'test/helpers/gemini-session-runner.ts',
];
for (const rel of runners) {
test(`${path.basename(rel)}: detached spawn + killProcessGroup, no bare timeout kill`, () => {
-233
View File
@@ -1,233 +0,0 @@
/**
* Privacy-gate E2E (periodic tier, paid).
*
* The gbrain-sync preamble block instructs the model to fire a one-time
* AskUserQuestion when:
* - `BRAIN_SYNC: off` in the preamble echo (sync mode not on)
* - config `artifacts_sync_mode_prompted` is "false"
* - gbrain is detected on the host (binary on PATH or `gbrain doctor`
* --fast --json succeeds)
*
* This test stages all three conditions (via env + a fake `gbrain` binary
* on PATH), runs a cheap gstack skill through the Agent SDK, intercepts
* every tool use via canUseTool, and asserts: one of the AskUserQuestions
* fired by the preamble is the privacy gate with its distinctive prose
* and three options (full / artifacts-only / decline).
*
* Cost: ~$0.30-$0.50 per run. Periodic tier (EVALS=1 EVALS_TIER=periodic).
*
* See scripts/resolvers/preamble/generate-brain-sync-block.ts for the
* prose contract this test locks in.
*/
import { test, expect } from 'bun:test';
import { CAPTURE_MS } from './helpers/eval-budgets';
import { describeE2ETier } from './helpers/e2e-gate';
import * as fs from 'fs';
import * as os from 'os';
import * as path from 'path';
import { runAgentSdkTest, passThroughNonAskUserQuestion, resolveClaudeBinary } from './helpers/agent-sdk-runner';
const describeE2E = describeE2ETier('periodic');
describeE2E('gbrain-sync privacy gate fires once via preamble', () => {
test('gstack skill preamble fires the 3-option AskUserQuestion when gbrain is detected', async () => {
// Stage a fresh GSTACK_HOME with artifacts_sync_mode_prompted=false.
const gstackHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-gstack-'));
const fakeBinDir = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-bin-'));
// Fresh HOME with NO ~/.claude.json: on a machine where gbrain is
// registered type=http, the preamble's remote-mode detection reads the
// operator's ~/.claude.json and echoes "ARTIFACTS_SYNC: remote-mode" —
// and the local privacy gate legitimately never fires. An empty HOME
// makes the detection find nothing.
const tempHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-home-'));
// Seed the config so the gate's condition passes.
fs.writeFileSync(
path.join(gstackHome, 'config.yaml'),
'artifacts_sync_mode: off\nartifacts_sync_mode_prompted: false\n',
{ mode: 0o600 }
);
// Fake `gbrain` binary that makes the host-detection probe succeed.
// The preamble checks `gbrain doctor --fast --json` OR `which gbrain`.
// Either branch counts as "gbrain detected."
fs.writeFileSync(
path.join(fakeBinDir, 'gbrain'),
'#!/bin/bash\n' +
'case "$1" in\n' +
' doctor) echo \'{"status":"ok","schema_version":2}\' ; exit 0 ;;\n' +
' --version) echo "0.18.2" ; exit 0 ;;\n' +
' *) exit 0 ;;\n' +
'esac\n',
{ mode: 0o755 }
);
const askUserQuestions: Array<{ input: Record<string, unknown> }> = [];
const binary = resolveClaudeBinary();
// Per-test env, merged LAST by the hermetic env builder (safe post-v1.39:
// the runner always passes a COMPLETE hermetic env, so overrides can't
// break auth). Ambient process.env.GSTACK_HOME mutation does NOT work
// here — hermetic-env scrubs GSTACK_* and repoints GSTACK_HOME at its
// own singleton dir, so the staged config would never reach the child.
const childEnv = {
GSTACK_HOME: gstackHome,
HOME: tempHome,
PATH: `${fakeBinDir}:${process.env.PATH ?? '/usr/bin:/bin:/opt/homebrew/bin'}`,
};
try {
// Pick a small skill with the preamble and load it via Read to force
// the model to execute every preamble directive. A narrow "run /learn"
// prompt often gets reduced to a direct action, skipping the preamble
// gates. Mirror the plan-mode-no-op test pattern: ask the model to
// follow the skill's instructions in full.
const learnSkill = path.resolve(
import.meta.dir,
'..',
'learn',
'SKILL.md'
);
await runAgentSdkTest({
systemPrompt: { type: 'preset', preset: 'claude_code' },
userPrompt:
`Read the skill file at ${learnSkill} and follow its instructions from the top, including every preamble directive. Execute every bash block. If any AskUserQuestion fires, present it.`,
workingDirectory: gstackHome,
maxTurns: 10,
allowedTools: ['Read', 'Grep', 'Glob', 'Bash'],
env: childEnv,
...(binary ? { pathToClaudeCodeExecutable: binary } : {}),
canUseTool: async (toolName, input) => {
if (toolName === 'AskUserQuestion') {
askUserQuestions.push({ input });
// Auto-answer "Decline — keep everything local" (option C)
// so the skill can continue without actually turning on sync.
const q = (input.questions as Array<{
question: string;
options: Array<{ label: string }>;
}>)[0];
const decline =
q.options.find((o) => /decline|keep everything local|no thanks/i.test(o.label)) ??
q.options[q.options.length - 1]!;
return {
behavior: 'allow',
updatedInput: {
questions: input.questions,
answers: { [q.question]: decline.label },
},
};
}
return passThroughNonAskUserQuestion(toolName, input);
},
});
// Assertion 1: the privacy gate fired.
const privacyQuestions = askUserQuestions.filter((aq) => {
const qs = aq.input.questions as Array<{ question: string }>;
return qs.some(
(q) =>
/publish.*session memory|private github repo|gbrain indexes/i.test(q.question)
);
});
expect(privacyQuestions.length).toBeGreaterThanOrEqual(1);
// Assertion 2: the question has the three expected options.
const gate = privacyQuestions[0]!.input.questions as Array<{
question: string;
options: Array<{ label: string }>;
}>;
const labels = gate[0]!.options.map((o) => o.label.toLowerCase()).join(' | ');
// Full / artifacts-only / decline are the three canonical options.
expect(labels).toMatch(/everything|allowlisted|full/);
expect(labels).toMatch(/artifact/);
expect(labels).toMatch(/decline|local|no thanks/);
// Assertion 3: the gate should NOT fire twice in one run.
// (The preamble is supposed to be idempotent within a session.)
expect(privacyQuestions.length).toBe(1);
} finally {
fs.rmSync(gstackHome, { recursive: true, force: true });
fs.rmSync(fakeBinDir, { recursive: true, force: true });
fs.rmSync(tempHome, { recursive: true, force: true });
}
}, CAPTURE_MS);
test('privacy gate does NOT fire when artifacts_sync_mode_prompted is already true', async () => {
// Same staging, but prompted=true this time. Gate should be silent.
const gstackHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-off-'));
const fakeBinDir = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-off-bin-'));
// Fresh HOME without a .claude.json — same rationale as the first test:
// without it the operator's ~/.claude.json flips the preamble into
// remote-mode and this negative test passes vacuously.
const tempHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-off-home-'));
fs.writeFileSync(
path.join(gstackHome, 'config.yaml'),
'artifacts_sync_mode: off\nartifacts_sync_mode_prompted: true\n',
{ mode: 0o600 }
);
fs.writeFileSync(
path.join(fakeBinDir, 'gbrain'),
'#!/bin/bash\necho \'{"status":"ok"}\'\nexit 0\n',
{ mode: 0o755 }
);
const askUserQuestions: Array<{ input: Record<string, unknown> }> = [];
const binary = resolveClaudeBinary();
// Per-test env, merged LAST by the hermetic env builder (see note on the
// first test — ambient GSTACK_HOME mutation is scrubbed by hermetic-env).
const childEnv = {
GSTACK_HOME: gstackHome,
HOME: tempHome,
PATH: `${fakeBinDir}:${process.env.PATH ?? '/usr/bin:/bin:/opt/homebrew/bin'}`,
};
try {
await runAgentSdkTest({
systemPrompt: { type: 'preset', preset: 'claude_code' },
userPrompt:
'Run /learn with no arguments. Just report the learnings count.',
workingDirectory: gstackHome,
maxTurns: 4,
allowedTools: ['Read', 'Grep', 'Glob', 'Bash'],
env: childEnv,
...(binary ? { pathToClaudeCodeExecutable: binary } : {}),
canUseTool: async (toolName, input) => {
if (toolName === 'AskUserQuestion') {
askUserQuestions.push({ input });
// Pass through whatever the model asks; don't prefer anything.
const q = (input.questions as Array<{
question: string;
options: Array<{ label: string }>;
}>)[0];
return {
behavior: 'allow',
updatedInput: {
questions: input.questions,
answers: { [q.question]: q.options[0]!.label },
},
};
}
return passThroughNonAskUserQuestion(toolName, input);
},
});
// No AskUserQuestion should have matched the privacy gate's prose.
const privacyQuestions = askUserQuestions.filter((aq) => {
const qs = aq.input.questions as Array<{ question: string }>;
return qs.some(
(q) =>
/publish.*session memory|private github repo|gbrain indexes/i.test(q.question)
);
});
expect(privacyQuestions.length).toBe(0);
} finally {
fs.rmSync(gstackHome, { recursive: true, force: true });
fs.rmSync(fakeBinDir, { recursive: true, force: true });
fs.rmSync(tempHome, { recursive: true, force: true });
}
}, CAPTURE_MS);
});
-76
View File
@@ -1,76 +0,0 @@
/**
* Conductor → prose decision brief (periodic-tier, paid, real-PTY).
*
* Proves the end-to-end behavior: when CONDUCTOR_SESSION is signalled, a skill
* that hits a decision renders a PROSE decision brief and waits, instead of
* silently skipping the user.
*
* SCOPE — read before trusting this as the Conductor guard. This is END-TO-END
* BEHAVIOR coverage, NOT the discriminating Conductor guarantee:
* - The deterministic guard is test/question-preference-hook.test.ts
* ("Conductor prose redirect") — it sets process.env.CONDUCTOR_* and asserts
* the PreToolUse hook denies + redirects. That test CAN fail on unfixed code.
* - The PTY harness here cannot register `mcp__conductor__AskUserQuestion`, so
* it tests "native AUQ unavailable + Conductor signal → prose," NOT "the MCP
* variant exists and must not be called" (Codex #10). Under --disallowedTools
* a present-human interactive session already prose-falls-back, so this test
* is a smoke check that the Conductor path still produces a prose brief, not
* a proof that the Conductor signal (vs the generic fallback) drove it.
*
* Periodic tier: model-behavior, non-deterministic.
*/
import { test, expect } from 'bun:test';
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
import { describeE2ETier } from './helpers/e2e-gate';
import { runPlanSkillObservation } from './helpers/claude-pty-runner';
const describeE2E = describeE2ETier('periodic');
const FLAWED_PLAN = `# Plan: add a "developer-friendly" pricing tier
## Goal
Increase developer adoption.
## Premise
No tests mentioned, no rollout plan, no auth check on the upgrade endpoint.
Adds a Stripe tier, a React pricing page, a Postgres entitlements table, and a
Redis cache. The team "feels like" it should be cheaper; no developer was asked.
`;
describeE2E('Conductor renders decisions as prose (periodic)', () => {
test('plan-eng-review in a Conductor session surfaces a PROSE decision brief, not a silent skip', async () => {
const obs = await runPlanSkillObservation({
skillName: 'plan-eng-review',
inPlanMode: true,
// Mimic Conductor: native AUQ disabled + the Conductor env signal present.
extraArgs: ['--disallowedTools', 'AskUserQuestion'],
env: { CONDUCTOR_WORKSPACE_PATH: '/tmp/conductor-prose-e2e' },
initialPlanContent: FLAWED_PLAN,
// A judge can mistake a streaming partial option for a waiting brief.
// Keep observing until the independent prose evidence is available.
requireProseEvidence: true,
timeoutMs: CAPTURE_MS,
});
// The decision must reach the human as prose. 'silent_write' (wrote findings
// to the plan without asking) is the precise failure we guard against.
if (obs.outcome === 'silent_write') {
throw new Error(
`Conductor prose regression: skill wrote findings without surfacing a decision.\n` +
`summary: ${obs.summary}\n--- evidence ---\n${obs.evidence}`,
);
}
if (obs.outcome === 'exited' || obs.outcome === 'timeout') {
throw new Error(
`Conductor prose test inconclusive: outcome=${obs.outcome}\n` +
`summary: ${obs.summary}\n--- evidence ---\n${obs.evidence}`,
);
}
// A prose-rendered decision brief was observed at some point in the run.
expect(obs.proseAUQEverObserved,
`Conductor prose decision not observed: outcome=${obs.outcome}\n` +
`summary: ${obs.summary}\n--- evidence ---\n${obs.evidence}`,
).toBe(true);
}, CAPTURE_LONG_MS);
});
-268
View File
@@ -1,268 +0,0 @@
/**
* Opus 4.7 behavior evals.
*
* One case, pinned to claude-opus-4-7:
*
* Routing precision — the "when in doubt, invoke the skill" policy should
* route ambiguous dev prompts to the right skill WITHOUT routing
* casual/non-dev prompts. A handful of positive and negative controls.
* (The fanout A/B retired 2026-08 — single-run parallel-call comparison was
* a coin flip; the SDK overlay-harness is the maintained instrument.)
*
* Both cases require a running Anthropic API key. Gated behind EVALS=1.
* Classify as `periodic` in touchfiles — behavior measurement, not gate.
*/
import { describe, test, expect, afterAll } from 'bun:test';
import { JUDGE_MS, CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
import { runSkillTest } from './helpers/session-runner';
import { EvalCollector } from './helpers/eval-store';
import { extractSkillHead } from './helpers/skill-fixture';
import { spawnSync } from 'child_process';
import * as fs from 'fs';
import * as path from 'path';
import * as os from 'os';
const ROOT = path.resolve(import.meta.dir, '..');
const OPUS_47 = 'claude-opus-4-7';
const evalsEnabled = !!process.env.EVALS;
const describeE2E = evalsEnabled ? describe : describe.skip;
const evalCollector = evalsEnabled ? new EvalCollector('e2e-opus-47') : null;
const runId = new Date().toISOString().replace(/[:.]/g, '').replace('T', '-').slice(0, 15);
// --- Helpers ---
/** Skills that must exist as individual .claude/skills/{name}/SKILL.md files
* for Claude Code's auto-discovery to treat them as invokable via Skill tool.
* Matches the pattern in skill-routing-e2e.test.ts. */
const INSTALLED_SKILLS = [
'qa', 'qa-only', 'ship', 'review', 'plan-ceo-review', 'plan-eng-review',
'plan-design-review', 'design-review', 'design-consultation', 'retro',
'document-release', 'investigate', 'office-hours', 'browse',
];
/** Write a scratch root with:
* - Per-skill SKILL.md files under .claude/skills/ (so Skill tool sees them)
* - Project CLAUDE.md with explicit routing rules AND (optionally) the
* 4.7 overlay content directly inlined so `claude -p` sees it
* - git init
*
* `includeOverlay` controls whether the opus-4-7 nudges (Fan out, Literal,
* etc.) get inlined into CLAUDE.md — this is the A/B axis for the fanout
* test. `claude -p` doesn't auto-load SKILL.md content, so CLAUDE.md is
* the only way to make the overlay visible to the model in this test
* harness.
*/
function mkEvalRoot(suffix: string, includeOverlay: boolean): string {
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), `opus47-${suffix}-`));
// Render at opus-4-7 INTO A MKDTEMP via --out-dir — never the live tree.
// The previous cwd=ROOT regeneration rewrote every in-repo SKILL.md
// mid-run while concurrent paid shards copyFileSync those same files in
// their beforeAll (EVALS_JOBS>=4 locally, 2 per CI slice): a sibling could
// capture a half-regenerated or opus-rendered SKILL.md, and a timeout
// before afterAll left the whole tree rendered at the wrong model for
// every later shard. --out-dir mirrors the repo layout (<out>/<skill>/
// SKILL.md), which is all this fixture reads.
const renderDir = fs.mkdtempSync(path.join(os.tmpdir(), `opus47-render-${suffix}-`));
const result = spawnSync(
'bun',
['run', 'scripts/gen-skill-docs.ts', '--model', includeOverlay ? 'opus-4-7' : 'claude', '--out-dir', renderDir],
{ cwd: ROOT, stdio: 'pipe', encoding: 'utf-8', timeout: 60_000 },
);
if (result.status !== 0) {
throw new Error(`gen-skill-docs failed: ${result.stderr}`);
}
// Install per-skill SKILL.md files for Skill tool discovery. Routing only
// reads the frontmatter (name + description), so install frontmatter + the
// first ~30 body lines instead of the full 1000-1900-line files
// (CLAUDE.md: "E2E test fixtures: extract, don't copy").
const skillsDir = path.join(tmp, '.claude', 'skills');
for (const skill of INSTALLED_SKILLS) {
const src = path.join(renderDir, skill, 'SKILL.md');
if (!fs.existsSync(src)) continue;
const destDir = path.join(skillsDir, skill);
fs.mkdirSync(destDir, { recursive: true });
fs.writeFileSync(path.join(destDir, 'SKILL.md'), extractSkillHead(src));
}
fs.rmSync(renderDir, { recursive: true, force: true });
// Extract the opus-4-7 model-overlay content from the checked-in file
// so we can inline it into CLAUDE.md when includeOverlay is true.
const overlayText = includeOverlay
? fs.readFileSync(path.join(ROOT, 'model-overlays', 'opus-4-7.md'), 'utf-8')
.replace(/\{\{INHERIT:claude\}\}\s*/, '')
.trim()
: '';
// Project CLAUDE.md. Explicit routing rules so the agent reaches for
// Skill tool on matching prompts, plus the optional overlay.
const routingBlock = `## Skill routing
When the user's request matches an available skill, invoke it via the Skill tool
as your FIRST action. The skill has multi-step workflows, checklists, and quality
gates that produce better results than an ad-hoc answer. When in doubt, invoke.
- Bugs, errors, "why is this broken", "wtf" → invoke investigate
- Ship, deploy, "send it", create a PR → invoke ship
- QA, test the site, "does this work" → invoke qa
- Code review, check my diff → invoke review
- Product ideas, brainstorming, "is this worth building" → invoke office-hours
- Architecture, "does this design make sense" → invoke plan-eng-review
- Design system, visual polish → invoke design-review
- Weekly retro, what did we ship → invoke retro`;
const claudeMd = includeOverlay
? `# Project\n\n${overlayText}\n\n${routingBlock}\n`
: `# Project\n\n${routingBlock}\n`;
fs.writeFileSync(path.join(tmp, 'CLAUDE.md'), claudeMd);
fs.writeFileSync(path.join(tmp, 'package.json'), '{"name":"opus47-eval"}');
const git = (args: string[]) =>
spawnSync('git', args, { cwd: tmp, stdio: 'pipe', timeout: 5_000 });
git(['init']);
git(['config', 'user.email', 't@t.com']);
git(['config', 'user.name', 'T']);
git(['add', '.']);
git(['commit', '-m', 'init']);
return tmp;
}
/** Count parallel tool calls in the first assistant turn. */
function firstTurnParallelism(transcript: any[]): number {
const firstAssistant = transcript.find((e) => e.type === 'assistant');
if (!firstAssistant) return 0;
const content = firstAssistant.message?.content ?? [];
return content.filter((c: any) => c.type === 'tool_use').length;
}
interface RoutingCase {
name: string;
prompt: string;
shouldRoute: boolean;
expectedSkill?: string;
}
/** Small, intentionally chosen routing cases. Positive cases are ambiguous
* phrasings the user actually says, not template text. Negative cases are
* casual or off-topic prompts that match routing keywords but shouldn't
* trigger a skill. */
const ROUTING_CASES: RoutingCase[] = [
// Positive — should route
{ name: 'pos-wtf-bug', prompt: "wtf is this error coming from auth.ts:47 when the cookie expires?", shouldRoute: true, expectedSkill: 'investigate' },
{ name: 'pos-send-it', prompt: "ok this is good enough, let's send it.", shouldRoute: true, expectedSkill: 'ship' },
{ name: 'pos-does-it-work', prompt: "I just pushed the login flow changes. Test the deployed site and find any bugs.", shouldRoute: true, expectedSkill: 'qa' },
// Negative — should NOT route
{ name: 'neg-syntax-q', prompt: "wtf does this Python list comprehension syntax even mean, [x for x in y if z]?", shouldRoute: false },
{ name: 'neg-algo-q', prompt: "does this bubble sort algorithm actually work in O(n log n)?", shouldRoute: false },
{ name: 'neg-slack-send', prompt: "can you help me write the slack message? I want to send it to the team.", shouldRoute: false },
];
// --- Tests ---
describeE2E('Opus 4.7 overlay behavior evals', () => {
afterAll(() => {
evalCollector?.finalize();
// No tree restore needed: mkEvalRoot renders into a mkdtemp via
// --out-dir, so the live repo's SKILL.md files are never touched — a
// timeout mid-run can no longer strand the tree at the wrong model for
// concurrent shards.
});
// (fanout A/B retired, 2026-08 audit: it compared parallel-call counts of
// two SINGLE stochastic runs — parA >= parB is a coin flip with near-zero
// remaining information; the overlay-fanout question is answered and the
// SDK overlay-harness (test/skill-e2e-overlay-harness.test.ts) is the
// maintained instrument for the next experiment.)
test(
'routing precision: positives route, negatives do not',
async () => {
// Single SKILL.md tree shared by all cases. We run claude-opus-4-7 with
// tool access to Skill; measure whether the first tool call is Skill(..)
// and if so, which skill.
const root = mkEvalRoot('routing', true);
try {
const results = await Promise.all(
ROUTING_CASES.map((c) =>
runSkillTest({
prompt: c.prompt,
workingDirectory: root,
maxTurns: 3,
allowedTools: ['Skill', 'Read', 'Bash', 'Glob', 'Grep'],
timeout: JUDGE_MS,
testName: `routing-${c.name}`,
runId,
model: OPUS_47,
}).then((r) => ({ c, r })),
),
);
let tp = 0, fn = 0, fp = 0, tn = 0;
const rows: string[] = [];
let totalCost = 0;
for (const { c, r } of results) {
const skillCalls = r.toolCalls.filter((tc) => tc.tool === 'Skill');
const routed = skillCalls.length > 0;
const actualSkill = routed ? skillCalls[0]?.input?.skill : undefined;
const correct = c.shouldRoute
? routed && (!c.expectedSkill || actualSkill === c.expectedSkill)
: !routed;
if (c.shouldRoute && routed) tp++;
else if (c.shouldRoute && !routed) fn++;
else if (!c.shouldRoute && routed) fp++;
else tn++;
totalCost += r.costEstimate.estimatedCost;
rows.push(
` ${c.name.padEnd(18)} routed=${String(routed).padEnd(5)} skill=${String(actualSkill).padEnd(16)} ` +
`expected=${c.shouldRoute ? (c.expectedSkill ?? 'any') : '(none)'} ${correct ? 'OK' : 'MISS'}`,
);
evalCollector?.addTest({
name: `routing-${c.name}`,
suite: 'Opus 4.7 routing',
tier: 'e2e',
passed: correct,
duration_ms: r.duration,
cost_usd: r.costEstimate.estimatedCost,
transcript: r.transcript,
output: `routed=${routed} actual=${actualSkill ?? '(none)'} expected=${c.shouldRoute ? c.expectedSkill ?? 'any' : '(none)'}`,
turns_used: r.costEstimate.turnsUsed,
exit_reason: r.exitReason,
});
}
const posCount = ROUTING_CASES.filter((c) => c.shouldRoute).length;
const negCount = ROUTING_CASES.length - posCount;
const tpRate = posCount > 0 ? tp / posCount : 0;
const fpRate = negCount > 0 ? fp / negCount : 0;
console.log(`[opus-4-7 routing] total cost $${totalCost.toFixed(2)}`);
console.log(rows.join('\n'));
console.log(
` TP=${tp}/${posCount} (${(tpRate * 100).toFixed(0)}%) FN=${fn} ` +
`FP=${fp}/${negCount} (${(fpRate * 100).toFixed(0)}%) TN=${tn}`,
);
// Thresholds from the test plan artifact: TP >= 80%, FP <= 30%.
// With a small N we loosen slightly: TP >= 66% (2 of 3 positive),
// FP <= 33% (no more than 1 of 3 negatives).
expect(tpRate, `true-positive rate ${(tpRate * 100).toFixed(0)}% (need >= 66%)`).toBeGreaterThanOrEqual(2 / 3);
expect(fpRate, `false-positive rate ${(fpRate * 100).toFixed(0)}% (need <= 33%)`).toBeLessThanOrEqual(1 / 3);
} finally {
fs.rmSync(root, { recursive: true, force: true });
}
},
CAPTURE_LONG_MS,
);
});
@@ -1,6 +0,0 @@
import { describeE2ETier } from './helpers/e2e-gate';
import { registerOverlayCase } from './helpers/overlay-case';
describeE2ETier('periodic')('overlay behavior contract v2 (SDK)', () => {
registerOverlayCase('opus-4-7-effort-match-trivial-sonnet');
});
@@ -1,6 +0,0 @@
import { describeE2ETier } from './helpers/e2e-gate';
import { registerOverlayCase } from './helpers/overlay-case';
describeE2ETier('periodic')('overlay behavior contract v2 (SDK)', () => {
registerOverlayCase('opus-4-7-literal-interpretation-sonnet');
});
-285
View File
@@ -1,285 +0,0 @@
/**
* /ship idempotency E2E (periodic, paid, real-PTY).
*
* Asserts: when /ship runs against a branch that has ALREADY been bumped
* (VERSION ahead of base AND package.json synced AND a CHANGELOG entry
* exists for the bumped version), the workflow:
*
* 1. Detects ALREADY_BUMPED state via the Step 12 idempotency check
* 2. Does NOT echo STATE: FRESH (which would trigger a second bump)
* 3. Does NOT mutate the fixture's VERSION file
* 4. Does NOT append a duplicate CHANGELOG [0.0.2] entry
* 5. Does NOT create a new "chore: bump version" commit
*
* Why real-PTY: the old SDK-harness ship-idempotency variant (removed in
* v1.64.1.0 as redundant with this test) used a synthetic prompt asking
* the agent to "run ONLY the idempotency checks." This test exercises the
* actual /ship skill end-to-end against a real git fixture so a regression
* that silently re-bumps despite the check passing would be caught.
*
* Plan-mode framing: we run /ship in plan mode so the agent cannot push,
* commit, or open PRs. The Step 12 idempotency check is read-only
* (reads VERSION + package.json + git rev-parse) and runs fine in plan
* mode. The plan-ready output serves as the terminal signal — the agent
* has done its analysis and produced a plan describing what it would do.
*
* If the agent decides to bump or push despite the fixture's
* ALREADY_BUMPED state, that intent surfaces in the plan or in
* tool-call attempts, which we detect.
*
* Cost: ~$2-4/run. Periodic tier — long, runs weekly.
*/
import { test, expect } from 'bun:test';
import { PTY_LONG_MS } from './helpers/eval-budgets';
import { describeE2ETier } from './helpers/e2e-gate';
import { spawnSync } from 'child_process';
import * as fs from 'fs';
import * as path from 'path';
import * as os from 'os';
import {
launchClaudePty,
isPermissionDialogVisible,
isNumberedOptionListVisible,
} from './helpers/claude-pty-runner';
const describeE2E = describeE2ETier('periodic');
interface ShipFixture {
workTree: string;
bareRemote: string;
/** Full bash log of `git` and helper commands run during setup. */
setupLog: string[];
}
/**
* Build a self-contained git fixture representing an already-shipped state:
* - main branch at VERSION 0.0.1, with one CHANGELOG entry [0.0.1]
* - feat/already-shipped branch at VERSION 0.0.2 (bumped + synced),
* CHANGELOG has [0.0.2] entry on top of [0.0.1], one feature commit
* - bareRemote is the origin; both branches are pushed
*
* Returns the work-tree dir for /ship to operate on.
*/
function buildShippedFixture(): ShipFixture {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ship-fixture-'));
const workTree = path.join(root, 'workspace');
const bareRemote = path.join(root, 'origin.git');
fs.mkdirSync(workTree, { recursive: true });
const setupLog: string[] = [];
const sh = (cmd: string, cwd: string): void => {
setupLog.push(`[${cwd}] ${cmd}`);
const result = spawnSync('bash', ['-c', cmd], { cwd, stdio: 'pipe', timeout: 15_000 });
if (result.status !== 0) {
const stderr = result.stderr?.toString() ?? '';
throw new Error(`fixture setup failed at "${cmd}":\n${stderr}\n--- log ---\n${setupLog.join('\n')}`);
}
};
// Bare remote.
sh(`git init --bare "${bareRemote}"`, root);
// Initial commit on main.
sh('git init -b main', workTree);
sh('git config user.email "test@test.com"', workTree);
sh('git config user.name "Test"', workTree);
sh('git config commit.gpgsign false', workTree);
fs.writeFileSync(path.join(workTree, 'VERSION'), '0.0.1\n');
fs.writeFileSync(
path.join(workTree, 'package.json'),
JSON.stringify({ name: 'fixture', version: '0.0.1', private: true }, null, 2) + '\n',
);
fs.writeFileSync(
path.join(workTree, 'CHANGELOG.md'),
`# Changelog\n\n## [0.0.1] - 2026-01-01\n\n- Initial release\n`,
);
fs.writeFileSync(path.join(workTree, 'README.md'), '# Fixture\n');
sh('git add VERSION package.json CHANGELOG.md README.md', workTree);
sh('git commit -m "chore: initial release v0.0.1"', workTree);
sh(`git remote add origin "${bareRemote}"`, workTree);
sh('git push -u origin main', workTree);
// Feature branch with ALREADY_BUMPED state.
sh('git checkout -b feat/already-shipped', workTree);
fs.writeFileSync(path.join(workTree, 'VERSION'), '0.0.2\n');
fs.writeFileSync(
path.join(workTree, 'package.json'),
JSON.stringify({ name: 'fixture', version: '0.0.2', private: true }, null, 2) + '\n',
);
fs.writeFileSync(
path.join(workTree, 'CHANGELOG.md'),
`# Changelog\n\n## [0.0.2] - 2026-04-25\n\n**Feature shipped.**\n\nAdded the new feature.\n\n## [0.0.1] - 2026-01-01\n\n- Initial release\n`,
);
fs.writeFileSync(path.join(workTree, 'feature.md'), '# Feature\n\nAlready shipped.\n');
sh('git add VERSION package.json CHANGELOG.md feature.md', workTree);
sh('git commit -m "feat: add new feature\n\nbumps VERSION to 0.0.2"', workTree);
sh('git push -u origin feat/already-shipped', workTree);
return { workTree, bareRemote, setupLog };
}
/** Snapshot the load-bearing fixture state so we can compare post-run. */
interface FixtureSnapshot {
versionFile: string;
packageVersion: string;
changelogEntryCount: number;
bumpCommitCount: number;
branchHead: string;
}
function snapshotFixture(workTree: string): FixtureSnapshot {
const versionFile = fs.readFileSync(path.join(workTree, 'VERSION'), 'utf-8').trim();
const pkg = JSON.parse(fs.readFileSync(path.join(workTree, 'package.json'), 'utf-8'));
const changelog = fs.readFileSync(path.join(workTree, 'CHANGELOG.md'), 'utf-8');
// Count `## [0.0.2]` headings — should stay at 1 across re-runs.
const changelogEntryCount = (changelog.match(/^##\s*\[0\.0\.2\]/gm) ?? []).length;
const head = spawnSync('git', ['rev-parse', 'HEAD'], { cwd: workTree, stdio: 'pipe', timeout: 30_000 });
const branchHead = head.stdout?.toString().trim() ?? '';
// Count "chore: bump version" commits on this branch since main.
const log = spawnSync(
'git', ['log', '--format=%s', 'main..HEAD'],
{ cwd: workTree, stdio: 'pipe', timeout: 30_000 },
);
const subjects = log.stdout?.toString() ?? '';
const bumpCommitCount = subjects.split('\n').filter(s => /chore:\s*bump\s+version/i.test(s)).length;
return { versionFile, packageVersion: pkg.version, changelogEntryCount, bumpCommitCount, branchHead };
}
describeE2E('/ship idempotency E2E (periodic, real-PTY)', () => {
test(
'rerunning /ship on an already-shipped branch detects ALREADY_BUMPED and does not mutate fixture',
async () => {
const fixture = buildShippedFixture();
const before = snapshotFixture(fixture.workTree);
const session = await launchClaudePty({
permissionMode: 'plan',
cwd: fixture.workTree,
timeoutMs: PTY_LONG_MS,
// Disable network-y pieces so the agent can't reach actual github.
env: { GH_TOKEN: 'mock-not-real', NO_COLOR: '1' },
seedSkills: true,
});
let outcome: 'detected' | 'plan_ready' | 'attempted_mutation' | 'timeout' | 'exited' = 'timeout';
let evidence = '';
try {
await Bun.sleep(8000);
const since = session.mark();
session.send('/ship\r');
const budgetMs = 900_000;
const start = Date.now();
let lastPermSig = '';
while (Date.now() - start < budgetMs) {
await Bun.sleep(3000);
if (session.exited()) {
outcome = 'exited';
evidence = session.visibleSince(since).slice(-3000);
break;
}
const visible = session.visibleSince(since);
// Auto-grant any permission dialogs the preamble triggers
// (e.g. touch on a marker file claude considers sensitive).
// Classify on the recent tail; don't double-press the same render.
const tail = visible.slice(-1500);
if (isNumberedOptionListVisible(tail) && isPermissionDialogVisible(tail)) {
const sig = visible.slice(-500);
if (sig !== lastPermSig) {
lastPermSig = sig;
session.send('1\r');
await Bun.sleep(1500);
continue;
}
}
// Positive: idempotency classify reported ALREADY_BUMPED. Post-carve
// (T9), Step 12 runs `gstack-version-bump classify` which emits JSON
// (`"state":"ALREADY_BUMPED"`); the legacy inline bash echoed
// `STATE: ALREADY_BUMPED`. Accept either so the test survives the carve.
if (/STATE:\s*ALREADY_BUMPED|"state":\s*"ALREADY_BUMPED"/.test(visible)) {
outcome = 'detected';
evidence = visible.slice(-3000);
break;
}
// Negative regressions:
// - classify reported FRESH (CLI JSON or legacy echo) → would re-bump
// - agent attempted git commit -m "chore: bump version"
// - agent attempted git push
// - agent ran the CLI write path (gstack-version-bump write) — a
// re-bump on an already-shipped branch
if (
/"state":\s*"FRESH"/.test(visible) ||
/STATE:\s*FRESH(?![\w-])/i.test(visible) ||
/gstack-version-bump\s+write/i.test(visible) ||
/git\s+commit\s+.*chore:\s*bump\s+version/i.test(visible) ||
/git\s+push.*origin/i.test(visible)
) {
outcome = 'attempted_mutation';
evidence = visible.slice(-3000);
break;
}
// Plan-ready outcome (acceptable terminal): the agent finished
// analysis. We'll accept this if no mutation signals showed up.
if (/ready to execute|Would you like to proceed/i.test(visible)) {
outcome = 'plan_ready';
evidence = visible.slice(-3000);
break;
}
}
// Budget exhausted without a terminal signal: capture the tail NOW,
// while the session is still alive. Only the break paths above set
// evidence — without this, the timeout throw ships evidence: "".
if (outcome === 'timeout') {
evidence = session.visibleSince(since).slice(-3000);
}
} finally {
await session.close();
}
// Verify fixture was not mutated regardless of outcome.
const after = snapshotFixture(fixture.workTree);
const fixtureStable =
after.versionFile === before.versionFile &&
after.packageVersion === before.packageVersion &&
after.changelogEntryCount === before.changelogEntryCount &&
after.bumpCommitCount === before.bumpCommitCount &&
after.branchHead === before.branchHead;
try {
if (outcome === 'attempted_mutation') {
throw new Error(
`/ship attempted to mutate already-shipped state.\n` +
`--- evidence (last 3KB) ---\n${evidence}\n` +
`--- before ---\n${JSON.stringify(before, null, 2)}\n` +
`--- after ---\n${JSON.stringify(after, null, 2)}`,
);
}
if (outcome === 'exited') {
throw new Error(`claude exited unexpectedly.\n--- evidence ---\n${evidence}`);
}
if (outcome === 'timeout') {
throw new Error(
`Timed out before any terminal outcome.\n--- evidence (last 3KB) ---\n${evidence}`,
);
}
// Detected or plan_ready — both are acceptable terminal outcomes.
expect(['detected', 'plan_ready']).toContain(outcome);
// Fixture must not have been mutated regardless of outcome.
expect(fixtureStable).toBe(true);
} finally {
// Clean up fixture root.
try { fs.rmSync(path.dirname(fixture.workTree), { recursive: true, force: true }); } catch { /* ignore */ }
}
},
PTY_LONG_MS, // 20 min wall clock
);
});
-34
View File
@@ -1,34 +0,0 @@
/**
* /spec --execute end-to-end (periodic, paid, real-PTY).
*
* Asserts: when /spec --execute runs against a fixture prompt, it:
* 1. Refuses to draft on turn 1 (Phase 1 hard gate)
* 2. Reads code in Phase 3 (cites a real file path from the fixture repo)
* 3. Passes the quality gate (score >= 7) on a well-formed fixture
* 4. Spawns a fresh worktree on branch spec/<slug>-<pid>
* 5. Issues a final-confirm AskUserQuestion before the spawn
*
* Cost: ~$3-5/run, 5-8 min wall clock. Periodic — runs weekly via cron or
* on demand via `EVALS=1 EVALS_TIER=periodic bun run test:e2e`.
*
* TODO (v1.1): expand to test all 5 expansion paths and the plan-mode-aware
* Phase 5 branching (active vs inactive). Current implementation is the
* minimum smoke that proves --execute end-to-end works.
*/
import { test } from 'bun:test';
import { describeE2ETier } from './helpers/e2e-gate';
const describeE2E = describeE2ETier('periodic');
describeE2E('/spec --execute end-to-end (periodic)', () => {
// test.todo, not expect(true): the placeholder reported PASS on every
// periodic run while asserting nothing — a lying green with a 600s budget.
// The file itself stays: it is the periodic-tier surface registered in
// E2E_TIERS so the diff-based selector runs it when spec/ changes, and
// the deterministic template-invariant coverage in
// spec-template-invariants.test.ts + spec-template-sync.test.ts gates the
// gate tier. Implementation spec for the real PTY-driven test lives in
// the header TODO ("/spec --execute E2E full pipeline test (v1.1)").
test.todo('phase gating + magical Phase 3 + quality gate + spawn — full pipeline');
});
-35
View File
@@ -1,35 +0,0 @@
/**
* /spec LLM-judge eval (periodic, paid).
*
* Asserts: when /spec runs against a fixture vague request, the agent
* produces a spec body that scores >= 8/10 against an LLM judge using
* the contributor's 14 Quality Standards as the rubric.
*
* Cost: ~$0.15/run. Periodic — runs weekly via cron or on demand via
* `EVALS=1 EVALS_TIER=periodic bun run test:evals`.
*
* TODO (v1.1): expand fixture set to cover bug / feature / refactor / audit
* framings + project-level prompts (no concrete file mapping, exercises the
* Phase 3 fallback path).
*/
import { describe, test } from 'bun:test';
const evalsEnabled = !!process.env.EVALS;
const describeEval = evalsEnabled ? describe : describe.skip;
describeEval('/spec LLM-judge eval (periodic)', () => {
// test.todo, not expect(true): the placeholder reported PASS on every
// run while asserting nothing — a lying green with a 300s budget. The
// file stays as the periodic-tier selector surface for spec/ changes.
//
// Expected v1.1 implementation:
// 1. Pick fixture prompt from test/fixtures/spec/vague-bug.md
// 2. Spawn `claude -p` with /spec loaded, send the prompt + role-play
// five Phase 1 answers (from test/fixtures/spec/vague-bug-answers.json)
// 3. Capture final spec body
// 4. Dispatch to Claude judge with prompt encoding the 14 Quality
// Standards from spec/SKILL.md.tmpl
// 5. Assert numeric score >= 8
test.todo('spec body scores >= 8/10 against 14-standard rubric on fixture request');
});
+21 -199
View File
@@ -12,7 +12,6 @@
import { afterAll, expect } from 'bun:test';
import { JUDGE_MS } from './helpers/eval-budgets';
import Anthropic from '@anthropic-ai/sdk';
import * as fs from 'fs';
import * as path from 'path';
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
@@ -62,12 +61,11 @@ function readBrowseCommandSection(): string {
}
/** Slice a section out of the command-list section file, guarded non-empty. */
function sliceBrowseSection(startHeader: string, endHeader?: string): string {
function sliceBrowseSection(startHeader: string): string {
const content = readBrowseCommandSection();
const start = content.indexOf(startHeader);
if (start < 0) throw new Error(`browse/sections/command-list.md: "${startHeader}" not found`);
const end = endHeader ? content.indexOf(endHeader) : -1;
const section = end > start ? content.slice(start, end) : content.slice(start);
const section = content.slice(start);
if (section.trim().length < 200) {
throw new Error(`browse/sections/command-list.md slice at "${startHeader}" is empty/stub — regenerate with: bun run gen:skill-docs`);
}
@@ -91,85 +89,45 @@ function testIfSelected(testName: string, fn: () => Promise<void>, timeout: numb
}
describeIfSelected('LLM-as-judge quality evals', [
'command reference table', 'snapshot flags reference',
'browse/SKILL.md reference', 'setup block', 'regression vs baseline',
'browse/SKILL.md reference', 'setup block',
], () => {
testIfSelected('command reference table', async () => {
const t0 = Date.now();
// Browse carve: the command reference lives in the generated on-demand
// section browse/sections/command-list.md now (read via non-empty guard).
const section = sliceBrowseSection('## Full Command List');
const scores = await judge('command reference table', section);
console.log('Command reference scores:', JSON.stringify(scores, null, 2));
// Completeness threshold is 3 (not 4) — the command reference table is
// intentionally terse (quick-reference format). The judge consistently scores
// completeness=3 because detailed argument docs live in per-command sections.
evalCollector?.addTest({
name: 'command reference table',
suite: 'LLM-as-judge quality evals',
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
expect(scores.completeness).toBeGreaterThanOrEqual(3);
expect(scores.actionability).toBeGreaterThanOrEqual(4);
}, JUDGE_MS);
testIfSelected('snapshot flags reference', async () => {
const t0 = Date.now();
// Browse carve: snapshot flags live in browse/sections/command-list.md now,
// ordered before '## Full Command List' (the '## CSS Inspector' end boundary
// stayed in the skeleton).
const section = sliceBrowseSection('## Snapshot Flags', '## Full Command List');
const scores = await judge('snapshot flags reference', section);
console.log('Snapshot flags scores:', JSON.stringify(scores, null, 2));
evalCollector?.addTest({
name: 'snapshot flags reference',
suite: 'LLM-as-judge quality evals',
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
expect(scores.completeness).toBeGreaterThanOrEqual(4);
expect(scores.actionability).toBeGreaterThanOrEqual(4);
}, JUDGE_MS);
testIfSelected('browse/SKILL.md reference', async () => {
const t0 = Date.now();
// Browse carve: flags + commands are the whole generated section file.
// Browse carve: snapshot flags + the full command list are the whole
// generated section file; one judge grades the union. Scores are also
// pinned against test/fixtures/eval-baselines.json (UPDATE_BASELINES=1
// rewrites the pin).
const section = sliceBrowseSection('## Snapshot Flags');
const scores = await judge('browse skill reference (flags + commands)', section);
console.log('Browse SKILL.md scores:', JSON.stringify(scores, null, 2));
const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json');
const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8'));
const regressions = (['clarity', 'completeness', 'actionability'] as const)
.filter(dim => scores[dim] < baselines.browse_skill[dim])
.map(dim => `browse_skill.${dim}: ${scores[dim]} < baseline ${baselines.browse_skill[dim]}`);
if (process.env.UPDATE_BASELINES) {
baselines.browse_skill = { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability };
fs.writeFileSync(baselinesPath, JSON.stringify(baselines, null, 2) + '\n');
console.log('Updated eval baselines');
}
evalCollector?.addTest({
name: 'browse/SKILL.md reference',
suite: 'LLM-as-judge quality evals',
tier: 'llm-judge',
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4,
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4 && regressions.length === 0,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
judge_reasoning: scores.reasoning,
judge_reasoning: regressions.length ? `${scores.reasoning} | ${regressions.join('; ')}` : scores.reasoning,
});
expect(scores.clarity).toBeGreaterThanOrEqual(3);
expect(scores.completeness).toBeGreaterThanOrEqual(4);
expect(scores.actionability).toBeGreaterThanOrEqual(4);
expect(regressions).toEqual([]);
}, JUDGE_MS);
testIfSelected('setup block', async () => {
@@ -205,89 +163,6 @@ describeIfSelected('LLM-as-judge quality evals', [
expect(scores.clarity).toBeGreaterThanOrEqual(3);
}, JUDGE_MS);
testIfSelected('regression vs baseline', async () => {
const t0 = Date.now();
// Browse carve: the command reference lives in browse/sections/command-list.md.
const genSection = sliceBrowseSection('## Full Command List');
const baseline = `## Command Reference
### Navigation
| Command | Description |
|---------|-------------|
| \`goto <url>\` | Navigate to URL |
| \`back\` / \`forward\` | History navigation |
| \`reload\` | Reload page |
| \`url\` | Print current URL |
### Interaction
| Command | Description |
|---------|-------------|
| \`click <sel>\` | Click element |
| \`fill <sel> <val>\` | Fill input |
| \`select <sel> <val>\` | Select dropdown |
| \`hover <sel>\` | Hover element |
| \`type <text>\` | Type into focused element |
| \`press <key>\` | Press key (Enter, Tab, Escape) |
| \`scroll [sel]\` | Scroll element into view |
| \`wait <sel>\` | Wait for element (max 10s) |
| \`wait --networkidle\` | Wait for network to be idle |
| \`wait --load\` | Wait for page load event |
### Inspection
| Command | Description |
|---------|-------------|
| \`js <expr>\` | Run JavaScript |
| \`css <sel> <prop>\` | Computed CSS |
| \`attrs <sel>\` | Element attributes |
| \`is <prop> <sel>\` | State check (visible/hidden/enabled/disabled/checked/editable/focused) |
| \`console [--clear\\|--errors]\` | Console messages (--errors filters to error/warning) |`;
const client = new Anthropic();
const response = await client.messages.create({
model: 'claude-sonnet-4-6',
max_tokens: 1024,
messages: [{
role: 'user',
content: `You are comparing two versions of CLI documentation for an AI coding agent.
VERSION A (baseline — hand-maintained):
${baseline}
VERSION B (auto-generated from source):
${genSection}
Which version is better for an AI agent trying to use these commands? Consider:
- Completeness (more commands documented? all args shown?)
- Clarity (descriptions helpful?)
- Coverage (missing commands in either version?)
Respond with ONLY valid JSON:
{"winner": "A" or "B" or "tie", "reasoning": "brief explanation", "a_score": N, "b_score": N}
Scores are 1-5 overall quality.`,
}],
});
const text = response.content[0].type === 'text' ? response.content[0].text : '';
const jsonMatch = text.match(/\{[\s\S]*\}/);
if (!jsonMatch) throw new Error(`Judge returned non-JSON: ${text.slice(0, 200)}`);
const result = JSON.parse(jsonMatch[0]);
console.log('Regression comparison:', JSON.stringify(result, null, 2));
evalCollector?.addTest({
name: 'regression vs baseline',
suite: 'LLM-as-judge quality evals',
tier: 'llm-judge',
passed: result.b_score >= result.a_score,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
judge_scores: { a_score: result.a_score, b_score: result.b_score },
judge_reasoning: result.reasoning,
});
expect(result.b_score).toBeGreaterThanOrEqual(result.a_score);
}, JUDGE_MS);
});
// --- Part 7: QA skill quality evals (C6) ---
@@ -523,59 +398,6 @@ score (1-5): 5 = perfectly consistent, 1 = contradictory`);
}, JUDGE_MS);
});
// --- Part 7: Baseline score pinning (C9) ---
describeIfSelected('Baseline score pinning', ['baseline score pinning'], () => {
const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json');
testIfSelected('baseline score pinning', async () => {
const t0 = Date.now();
if (!fs.existsSync(baselinesPath)) {
console.log('No baseline file found — skipping pinning check');
return;
}
const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8'));
const regressions: string[] = [];
// Browse carve: the command reference lives in browse/sections/command-list.md.
const cmdSection = sliceBrowseSection('## Full Command List');
const cmdScores = await judge('command reference table', cmdSection);
for (const dim of ['clarity', 'completeness', 'actionability'] as const) {
if (cmdScores[dim] < baselines.command_reference[dim]) {
regressions.push(`command_reference.${dim}: ${cmdScores[dim]} < baseline ${baselines.command_reference[dim]}`);
}
}
if (process.env.UPDATE_BASELINES) {
baselines.command_reference = {
clarity: cmdScores.clarity,
completeness: cmdScores.completeness,
actionability: cmdScores.actionability,
};
fs.writeFileSync(baselinesPath, JSON.stringify(baselines, null, 2) + '\n');
console.log('Updated eval baselines');
}
const passed = regressions.length === 0;
evalCollector?.addTest({
name: 'baseline score pinning',
suite: 'Baseline score pinning',
tier: 'llm-judge',
passed,
duration_ms: Date.now() - t0,
cost_usd: 0.02,
judge_scores: { clarity: cmdScores.clarity, completeness: cmdScores.completeness, actionability: cmdScores.actionability },
judge_reasoning: passed ? 'All scores at or above baseline' : regressions.join('; '),
});
if (!passed) {
throw new Error(`Score regressions detected:\n${regressions.join('\n')}`);
}
}, JUDGE_MS);
});
// --- Workflow SKILL.md quality evals (10 new tests for 100% coverage) ---
/**
+46
View File
@@ -590,6 +590,52 @@ export default app;
}
}, CAPTURE_MS);
testIfSelected('journey-negatives', async () => {
// Casual or off-topic prompts that share routing keywords ("wtf",
// "algorithm", "send it to the team") must not invoke a skill. Folded
// from the retired Opus 4.7 routing eval; same bound: at most one of the
// three may route.
const cases = [
{ name: 'neg-syntax-q', prompt: 'wtf does this Python list comprehension syntax even mean, [x for x in y if z]?' },
{ name: 'neg-algo-q', prompt: 'does this bubble sort algorithm actually work in O(n log n)?' },
{ name: 'neg-slack-send', prompt: 'can you help me write the slack message? I want to send it to the team.' },
];
const tmpDir = createRoutingWorkDir('negatives');
try {
const results = await Promise.all(cases.map(async c => {
const result = await runSkillTest({
prompt: c.prompt,
workingDirectory: tmpDir,
maxTurns: 2,
allowedTools: ['Skill', 'Read'],
timeout: JUDGE_MS,
testName: `journey-negatives-${c.name}`,
runId,
});
const skillCalls = result.toolCalls.filter(tc => tc.tool === 'Skill');
const actualSkill = skillCalls.length > 0 ? skillCalls[0]?.input?.skill : undefined;
logCost(`journey: journey-negatives ${c.name}`, result);
evalCollector?.addTest({
name: `journey-negatives-${c.name}`,
suite: 'Skill Routing E2E',
tier: 'e2e',
passed: actualSkill === undefined,
duration_ms: result.duration,
cost_usd: result.costEstimate.estimatedCost,
transcript: result.transcript,
output: `routed=${actualSkill ?? '(none)'}`,
turns_used: result.costEstimate.turnsUsed,
exit_reason: result.exitReason,
});
return { name: c.name, actualSkill };
}));
const routed = results.filter(r => r.actualSkill !== undefined);
expect(routed.length, `negatives routed: ${routed.map(r => `${r.name}→${r.actualSkill}`).join(', ')}`).toBeLessThanOrEqual(1);
} finally {
fs.rmSync(tmpDir, { recursive: true, force: true });
}
}, CAPTURE_MS);
testIfSelected('journey-visual-qa', async () => {
const tmpDir = createRoutingWorkDir('visual-qa');
try {
-1
View File
@@ -181,7 +181,6 @@ describe('test-free-shards: enumeration', () => {
expect(isFreeTestFile('test/skill-llm-eval.test.ts')).toBe(false);
expect(isFreeTestFile('test/codex-e2e.test.ts')).toBe(false);
expect(isFreeTestFile('test/codex-e2e-sol-scope.test.ts')).toBe(false);
expect(isFreeTestFile('test/gemini-e2e.test.ts')).toBe(false);
});
test('collectFreeTestFiles returns sorted, deduped, only-free list', () => {
+6 -14
View File
@@ -83,8 +83,7 @@ describe('selectTests', () => {
// These cases use an outside-only/Code Quality excerpt or descriptive metadata.
.filter(id => !['outside-plan-disabled-no-fallback', 'plan-ceo-review-prosons-cadence',
'plan-review-prosons-format', 'shared-libs-plan-callers'].includes(id));
const existing = ['learnings-show', 'codex-plan-ceo-format-mode', 'codex-plan-ceo-format-approach',
'codex-plan-eng-format-coverage', 'codex-plan-eng-format-kind'];
const existing = ['learnings-show'];
const result = selectTests(['scripts/resolvers/learnings.ts'], E2E_TOUCHFILES);
expect(result.reason).toBe('diff');
expect(result.selected.sort()).toEqual([...new Set([...consumers, ...existing])].sort());
@@ -161,12 +160,8 @@ describe('selectTests', () => {
expect(fs.readFileSync(path.join(ROOT, `${output}.tmpl`), 'utf8')).toContain(`{{${token}}}`);
}
const generated = selectTests(consumers.map(([output]) => output), E2E_TOUCHFILES);
// These two CEO-format cases already depend on every resolver through
// scripts/resolvers/**; keep that existing selection alongside consumers.
// The bounded Code Quality fixture stops before Test review.
const expected = [...new Set([...generated.selected.filter(id => id !== 'shared-libs-plan-callers' && !SHIP_GUARD_ONLY.includes(id)),
'codex-plan-ceo-format-mode', 'codex-plan-ceo-format-approach',
])].sort();
const expected = [...new Set(generated.selected.filter(id => id !== 'shared-libs-plan-callers' && !SHIP_GUARD_ONLY.includes(id)))].sort();
const actual = selectTests(['scripts/resolvers/testing.ts'], E2E_TOUCHFILES);
expect(actual.reason).toBe('diff');
expect(actual.selected.sort()).toEqual(expected);
@@ -373,11 +368,9 @@ describe('selectTests', () => {
expect(result.selected).toContain('plan-ceo-split-overflow');
// v2 plan Phase B carve: the section-loading E2E depends on plan-ceo-review/**.
expect(result.selected).toContain('plan-ceo-section-loading');
expect(result.selected).toContain('codex-plan-ceo-format-mode');
expect(result.selected).toContain('codex-plan-ceo-format-approach');
expect(result.selected).toContain('outside-plan-disabled-no-fallback');
expect(result.selected.length).toBe(25);
expect(result.skipped.length).toBe(Object.keys(E2E_TOUCHFILES).length - 25);
expect(result.selected.length).toBe(23);
expect(result.skipped.length).toBe(Object.keys(E2E_TOUCHFILES).length - 23);
});
test('global touchfile triggers ALL tests', () => {
@@ -398,7 +391,6 @@ describe('selectTests', () => {
expect(result.reason).toBe('diff');
expect(result.selected).toContain('autoplan-chain-pty');
expect(result.selected).toContain('plan-ceo-mode-routing');
expect(result.selected).not.toContain('codex-plan-ceo-format-mode');
expect(result.selected).not.toContain('retro');
});
@@ -604,7 +596,7 @@ describe('TOUCHFILES completeness', () => {
);
const unique = registeredJudgeTestNames(llmContent);
expect(unique).toHaveLength(27);
expect(unique).toHaveLength(23);
const missing = unique.filter(name => !(name in LLM_JUDGE_TOUCHFILES));
if (missing.length > 0) {
@@ -623,7 +615,7 @@ describe('TOUCHFILES completeness', () => {
testIfSelected('unmapped judge case', async () => {}, 120_000);
`;
const names = registeredJudgeTestNames(withUnmappedCase);
expect(names).toHaveLength(28);
expect(names).toHaveLength(24);
expect(names.filter(name => !(name in LLM_JUDGE_TOUCHFILES))).toEqual(['unmapped judge case']);
});
+1 -1
View File
@@ -163,7 +163,7 @@ test('workflow registration preserves model work and reserves only terminal-reco
expect(source).toContain('const WORKFLOW_JUDGE_RECORD_MS = 5_000;');
expect(source).toContain('const WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000;');
expect(source.match(/\}, WORKFLOW_JUDGE_TEST_MS\);/g)).toHaveLength(16);
expect(source.match(/\}, JUDGE_MS\);/g)).toHaveLength(11);
expect(source.match(/\}, JUDGE_MS\);/g)).toHaveLength(7);
});
function actualCallback(f: ReturnType<typeof fixture>, overrides: {