mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
test: clean up the paid eval lane (B1-B4, B6, B7)
- B1: delete paid files that assert nothing or cannot pass meaningfully: skill-llm-eval-spec and skill-e2e-spec-execute (test.todo), gemini-e2e (+ gemini-session-runner; no gemini CLI in CI), ship-idempotency (red since v1.63), the two opus-4-7 *-sonnet overlay wrappers, conductor-prose (+ its source-evaluation replay), codex-e2e-plan-format; drop their keys, scripts and census rows. - B2: skill-llm-eval grades browse/sections/command-list.md with one union judge that also carries the baseline score pin; regression-vs-baseline deleted (paid run: pass, c4/c4/a4). - B3: memory-pipeline, ios-qa, ios-qa-swift-build and plan-tune-cathedral make no model calls; renamed out of the paid glob so they run on every PR. Swift builds need GSTACK_TEST_SWIFT=1; device stub deleted. - B4: codex-e2e*, outside-voice, aside and ios-device cannot run in the CI image; excluded from the weekly lane with a tracked re-entry condition. - B6: fold opus-47's negative routing controls into skill-routing-e2e journey-negatives (paid run: 3/3 unrouted) and delete the file. - B7: delete the never-green brain-privacy-gate eval; a free gstack-skill-start test now proves consent precedes artifacts egress.
This commit is contained in:
1 parent
5d032ef299
commit
53e7f3212f
58 files changed
+273
-2594
No files matched your search
@@ -50,7 +50,7 @@ test('failed loads, foreign sessions, actual questions and later withdrawals ret
|
||||
|
||||
import { E2E_TOUCHFILES } from './helpers/touchfiles-data';
|
||||
test('new evidence inputs retain every native observation caller',()=>{
|
||||
const owners=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','auto-decide-preserved','conductor-prose'];
|
||||
const owners=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','auto-decide-preserved'];
|
||||
for(const file of ['test/auto-decide-saved-ai.test.ts','test/fixtures/auto-decide-saved-ai.json','test/fixtures/auto-decide-retry-ai.json'])
|
||||
expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(file)).map(([name])=>name)).toEqual(owners);
|
||||
});
|
||||
|
||||
@@ -127,7 +127,7 @@ test('phase ordering and duplicate collapse use native time rather than polling
|
||||
|
||||
test('public narration changes select every existing shared native-reader consumer',()=>{
|
||||
const expected=[
|
||||
'auto-decide-preserved','autoplan-chain-pty','conductor-prose',
|
||||
'auto-decide-preserved','autoplan-chain-pty',
|
||||
'plan-ceo-finding-count','plan-ceo-mode-routing','plan-ceo-split-overflow',
|
||||
'plan-design-finding-count','plan-design-review-plan-mode','plan-design-with-ui-scope',
|
||||
'plan-devex-finding-count','plan-eng-finding-count','plan-eng-multi-finding-batching',
|
||||
|
||||
@@ -1,289 +0,0 @@
|
||||
/**
|
||||
* AskUserQuestion format regression test for /plan-ceo-review and /plan-eng-review
|
||||
* running under Codex CLI (GPT-5.4).
|
||||
*
|
||||
* Context: GPT-class models under the "No preamble / Prefer doing over listing"
|
||||
* gpt.md overlay tend to skip the Simplify (ELI10) paragraph and the RECOMMENDATION
|
||||
* line on AskUserQuestion calls. The user has to manually re-prompt "ELI10 and don't
|
||||
* forget to recommend" almost every time. This test pins that behavior so future
|
||||
* regressions surface automatically.
|
||||
*
|
||||
* Mirrors test/skill-e2e-plan-format.test.ts (the Claude version) but uses
|
||||
* test/helpers/codex-session-runner.ts to drive `codex exec` instead of `claude -p`.
|
||||
*
|
||||
* Four cases:
|
||||
* 1. plan-ceo-review mode selection (kind-differentiated)
|
||||
* 2. plan-ceo-review approach menu (coverage-differentiated)
|
||||
* 3. plan-eng-review per-issue coverage decision
|
||||
* 4. plan-eng-review per-issue architectural choice (kind-differentiated)
|
||||
*
|
||||
* Assertions on captured AskUserQuestion text:
|
||||
* - RECOMMENDATION: Choose present (all cases)
|
||||
* - Completeness: N/10 present on coverage cases, absent on kind cases
|
||||
* - "options differ in kind" note present on kind cases
|
||||
* - ELI10-style plain-English explanation present (length floor + no raw jargon)
|
||||
*
|
||||
* Periodic tier (Codex non-determinism). Cost: ~$2-3 per full run.
|
||||
*/
|
||||
import { describe, test, beforeAll, afterAll } from 'bun:test';
|
||||
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||
import { runCodexSkill } from './helpers/codex-session-runner';
|
||||
import { CODEX_EVAL_FINALIZE_MS, createCodexEvalCollector, runRecordedCodexEval, createCodexPlanFormatCapture } from './helpers/codex-eval';
|
||||
import { selectTests, detectBaseBranch, getChangedFiles, E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from './helpers/touchfiles';
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import * as os from 'os';
|
||||
import { spawnSync } from 'child_process';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
|
||||
// --- Prerequisites ---
|
||||
|
||||
const CODEX_AVAILABLE = (() => {
|
||||
try {
|
||||
const result = Bun.spawnSync(['which', 'codex'], { timeout: 30_000 });
|
||||
return result.exitCode === 0;
|
||||
} catch { return false; }
|
||||
})();
|
||||
const evalsEnabled = !!process.env.EVALS;
|
||||
// External-service test — periodic tier only (CLAUDE.md tiering rule 3),
|
||||
// matching codex-e2e.test.ts / codex-e2e-sol-scope.test.ts. Without this
|
||||
// guard the sharded runner's "no whole-file tier guard" default would run
|
||||
// Codex spawns in the GATE tier on every PR.
|
||||
const tierOk = process.env.EVALS_TIER === 'periodic';
|
||||
const SKIP = !CODEX_AVAILABLE || !evalsEnabled || !tierOk;
|
||||
const describeCodex = SKIP ? describe.skip : describe;
|
||||
|
||||
// --- Touchfiles ---
|
||||
|
||||
// Keep selection dependencies in the canonical map, including the test helpers.
|
||||
const CODEX_FORMAT_TOUCHFILES: Record<string, string[]> = Object.fromEntries(
|
||||
['codex-plan-ceo-format-mode', 'codex-plan-ceo-format-approach',
|
||||
'codex-plan-eng-format-coverage', 'codex-plan-eng-format-kind'].map((key) => {
|
||||
if (!E2E_TOUCHFILES[key]) throw new Error(`canonical E2E_TOUCHFILES lost key '${key}'`);
|
||||
return [key, E2E_TOUCHFILES[key]];
|
||||
}),
|
||||
);
|
||||
|
||||
let selectedTests: string[] | null = null;
|
||||
if (evalsEnabled && !process.env.EVALS_ALL) {
|
||||
const baseBranch = process.env.EVALS_BASE || detectBaseBranch(ROOT) || 'main';
|
||||
const changedFiles = getChangedFiles(baseBranch, ROOT);
|
||||
if (changedFiles.length > 0) {
|
||||
const selection = selectTests(changedFiles, CODEX_FORMAT_TOUCHFILES, GLOBAL_TOUCHFILES);
|
||||
selectedTests = selection.selected;
|
||||
}
|
||||
}
|
||||
|
||||
function testIfSelected(name: string, fn: () => Promise<void>, timeout: number) {
|
||||
if (selectedTests !== null && !selectedTests.includes(name)) {
|
||||
test.skip(name, fn, timeout + CODEX_EVAL_FINALIZE_MS);
|
||||
} else {
|
||||
test(name, fn, timeout + CODEX_EVAL_FINALIZE_MS);
|
||||
}
|
||||
}
|
||||
|
||||
// --- Eval collector ---
|
||||
|
||||
const evalCollector = SKIP ? null : createCodexEvalCollector('codex-e2e-plan-format');
|
||||
|
||||
afterAll(async () => {
|
||||
if (evalCollector) {
|
||||
await evalCollector.finalize();
|
||||
}
|
||||
});
|
||||
|
||||
// --- Fixtures ---
|
||||
|
||||
const SAMPLE_PLAN = `# Plan: Add User Dashboard
|
||||
|
||||
## Context
|
||||
We're building a new user dashboard that shows recent activity, notifications, and quick actions.
|
||||
|
||||
## Changes
|
||||
1. New React component \`UserDashboard\` in \`src/components/\`
|
||||
2. REST API endpoint \`GET /api/dashboard\` returning user stats
|
||||
3. PostgreSQL query for activity aggregation
|
||||
4. Redis cache layer for dashboard data (5min TTL)
|
||||
|
||||
## Architecture
|
||||
- Frontend: React + TailwindCSS
|
||||
- Backend: Express.js REST API
|
||||
- Database: PostgreSQL with existing user/activity tables
|
||||
- Cache: Redis for dashboard aggregates
|
||||
`;
|
||||
|
||||
function setupCodexSkillDir(tmpPrefix: string, skillName: 'plan-ceo-review' | 'plan-eng-review'): { skillDir: string; planDir: string; outFile: string } {
|
||||
const planDir = fs.mkdtempSync(path.join(os.tmpdir(), tmpPrefix));
|
||||
const run = (cmd: string, args: string[]) =>
|
||||
spawnSync(cmd, args, { cwd: planDir, stdio: 'pipe', timeout: 5000 });
|
||||
|
||||
run('git', ['init', '-b', 'main']);
|
||||
run('git', ['config', 'user.email', 'test@test.com']);
|
||||
run('git', ['config', 'user.name', 'Test']);
|
||||
|
||||
fs.writeFileSync(path.join(planDir, 'plan.md'), SAMPLE_PLAN);
|
||||
run('git', ['add', '.']);
|
||||
run('git', ['commit', '-m', 'add plan']);
|
||||
|
||||
// Codex skill lives in .agents/skills/gstack-{name}/ per the gstack host convention.
|
||||
const codexSkillSource = path.join(ROOT, '.agents', 'skills', `gstack-${skillName}`);
|
||||
const skillDir = path.join(planDir, '.agents', 'skills', `gstack-${skillName}`);
|
||||
fs.mkdirSync(skillDir, { recursive: true });
|
||||
fs.cpSync(codexSkillSource, skillDir, { recursive: true });
|
||||
|
||||
const outFile = path.join(planDir, 'ask-capture.md');
|
||||
return { skillDir, planDir, outFile };
|
||||
}
|
||||
|
||||
// Capture instruction — same shape as the Claude version. Codex may ignore tool calls,
|
||||
// so we tell it to write prose to the file directly.
|
||||
function captureInstruction(outFile: string): string {
|
||||
return `Write the verbatim text of every AskUserQuestion you would have presented to the user to the file ${outFile} (one question per session, full text including the re-ground, ELI10 paragraph, RECOMMENDATION line, and options). Do NOT ask the user interactively. Do NOT paraphrase. This is a format-capture test, not an interactive session.`;
|
||||
}
|
||||
|
||||
// --- Tests ---
|
||||
|
||||
describeCodex('Codex Plan Format — CEO Mode Selection', () => {
|
||||
let skillDir: string, planDir: string, outFile: string;
|
||||
|
||||
beforeAll(() => {
|
||||
({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-ceo-mode-', 'plan-ceo-review'));
|
||||
});
|
||||
|
||||
afterAll(() => {
|
||||
try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {}
|
||||
});
|
||||
|
||||
testIfSelected('codex-plan-ceo-format-mode', async () => {
|
||||
const capture = createCodexPlanFormatCapture(outFile, 'kind');
|
||||
const result = await runRecordedCodexEval({
|
||||
name: 'codex-plan-ceo-format-mode',
|
||||
suite: 'codex-e2e-plan-format',
|
||||
budgetMs: CAPTURE_LONG_MS,
|
||||
run: (signal) => {
|
||||
capture.reset();
|
||||
return runCodexSkill({
|
||||
skillDir,
|
||||
prompt: `Read the plan-ceo-review skill. Read plan.md (the plan to review). Proceed to Mode Selection where the skill presents 4 mode options (SCOPE EXPANSION, SELECTIVE EXPANSION, HOLD SCOPE, SCOPE REDUCTION) via AskUserQuestion. These options differ in kind (review posture), not coverage. ${captureInstruction(outFile)}`,
|
||||
timeoutMs: CAPTURE_MS,
|
||||
cwd: planDir,
|
||||
skillName: 'gstack-plan-ceo-review',
|
||||
sandbox: 'workspace-write',
|
||||
signal,
|
||||
});
|
||||
},
|
||||
validate: capture.validate,
|
||||
record: (entry) => evalCollector?.addTest(capture.attach(entry)),
|
||||
});
|
||||
console.log(`codex-plan-ceo-format-mode: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`);
|
||||
}, CAPTURE_LONG_MS);
|
||||
});
|
||||
|
||||
describeCodex('Codex Plan Format — CEO Approach Menu', () => {
|
||||
let skillDir: string, planDir: string, outFile: string;
|
||||
|
||||
beforeAll(() => {
|
||||
({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-ceo-approach-', 'plan-ceo-review'));
|
||||
});
|
||||
|
||||
afterAll(() => {
|
||||
try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {}
|
||||
});
|
||||
|
||||
testIfSelected('codex-plan-ceo-format-approach', async () => {
|
||||
const capture = createCodexPlanFormatCapture(outFile, 'coverage');
|
||||
const result = await runRecordedCodexEval({
|
||||
name: 'codex-plan-ceo-format-approach',
|
||||
suite: 'codex-e2e-plan-format',
|
||||
budgetMs: CAPTURE_LONG_MS,
|
||||
run: (signal) => {
|
||||
capture.reset();
|
||||
return runCodexSkill({
|
||||
skillDir,
|
||||
prompt: `Read the plan-ceo-review skill. Read plan.md. Proceed to Alternatives (the implementation approach menu) where the skill generates 2-3 approaches (minimal viable vs ideal architecture) and presents them via AskUserQuestion. These options differ in coverage so Completeness: N/10 applies. ${captureInstruction(outFile)}`,
|
||||
timeoutMs: CAPTURE_MS,
|
||||
cwd: planDir,
|
||||
skillName: 'gstack-plan-ceo-review',
|
||||
sandbox: 'workspace-write',
|
||||
signal,
|
||||
});
|
||||
},
|
||||
validate: capture.validate,
|
||||
record: (entry) => evalCollector?.addTest(capture.attach(entry)),
|
||||
});
|
||||
console.log(`codex-plan-ceo-format-approach: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`);
|
||||
}, CAPTURE_LONG_MS);
|
||||
});
|
||||
|
||||
describeCodex('Codex Plan Format — Eng Coverage Issue', () => {
|
||||
let skillDir: string, planDir: string, outFile: string;
|
||||
|
||||
beforeAll(() => {
|
||||
({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-eng-cov-', 'plan-eng-review'));
|
||||
});
|
||||
|
||||
afterAll(() => {
|
||||
try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {}
|
||||
});
|
||||
|
||||
testIfSelected('codex-plan-eng-format-coverage', async () => {
|
||||
const capture = createCodexPlanFormatCapture(outFile, 'coverage');
|
||||
const result = await runRecordedCodexEval({
|
||||
name: 'codex-plan-eng-format-coverage',
|
||||
suite: 'codex-e2e-plan-format',
|
||||
budgetMs: CAPTURE_LONG_MS,
|
||||
run: (signal) => {
|
||||
capture.reset();
|
||||
return runCodexSkill({
|
||||
skillDir,
|
||||
prompt: `Read the plan-eng-review skill. Read plan.md. In your Section 3 Test Review, generate ONE AskUserQuestion about test coverage depth where options are clearly coverage-differentiated: A) full coverage incl. edge + error paths (Completeness 10/10), B) happy path only (7/10), C) smoke test (3/10). ${captureInstruction(outFile)}`,
|
||||
timeoutMs: CAPTURE_MS,
|
||||
cwd: planDir,
|
||||
skillName: 'gstack-plan-eng-review',
|
||||
sandbox: 'workspace-write',
|
||||
signal,
|
||||
});
|
||||
},
|
||||
validate: capture.validate,
|
||||
record: (entry) => evalCollector?.addTest(capture.attach(entry)),
|
||||
});
|
||||
console.log(`codex-plan-eng-format-coverage: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`);
|
||||
}, CAPTURE_LONG_MS);
|
||||
});
|
||||
|
||||
describeCodex('Codex Plan Format — Eng Kind Issue', () => {
|
||||
let skillDir: string, planDir: string, outFile: string;
|
||||
|
||||
beforeAll(() => {
|
||||
({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-eng-kind-', 'plan-eng-review'));
|
||||
});
|
||||
|
||||
afterAll(() => {
|
||||
try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {}
|
||||
});
|
||||
|
||||
testIfSelected('codex-plan-eng-format-kind', async () => {
|
||||
const capture = createCodexPlanFormatCapture(outFile, 'kind');
|
||||
const result = await runRecordedCodexEval({
|
||||
name: 'codex-plan-eng-format-kind',
|
||||
suite: 'codex-e2e-plan-format',
|
||||
budgetMs: CAPTURE_LONG_MS,
|
||||
run: (signal) => {
|
||||
capture.reset();
|
||||
return runCodexSkill({
|
||||
skillDir,
|
||||
prompt: `Read the plan-eng-review skill. Read plan.md. In your Section 1 Architecture review, generate ONE AskUserQuestion about an architectural choice where the options differ in kind (e.g. Redis vs Postgres materialized view vs in-process cache — different kinds of systems with different tradeoffs, NOT more-or-less-complete versions of the same thing). ${captureInstruction(outFile)}`,
|
||||
timeoutMs: CAPTURE_MS,
|
||||
cwd: planDir,
|
||||
skillName: 'gstack-plan-eng-review',
|
||||
sandbox: 'workspace-write',
|
||||
signal,
|
||||
});
|
||||
},
|
||||
validate: capture.validate,
|
||||
record: (entry) => evalCollector?.addTest(capture.attach(entry)),
|
||||
});
|
||||
console.log(`codex-plan-eng-format-kind: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`);
|
||||
}, CAPTURE_LONG_MS);
|
||||
});
|
||||
@@ -11,23 +11,11 @@ describe('Codex eval selection', () => {
|
||||
expect(selected.sort()).toEqual([
|
||||
'codex-discover-skill',
|
||||
'codex-review-findings',
|
||||
'codex-plan-ceo-format-mode',
|
||||
'codex-plan-ceo-format-approach',
|
||||
'codex-plan-eng-format-coverage',
|
||||
'codex-plan-eng-format-kind',
|
||||
'codex-sol-scope-termination',
|
||||
].sort());
|
||||
expect(selected.every((id) => E2E_TIERS[id] === 'periodic')).toBe(true);
|
||||
});
|
||||
|
||||
test('format cases are selected by canonical source and their own test file', () => {
|
||||
expect(selectedBy('plan-ceo-review/SKILL.md.tmpl')).toContain('codex-plan-ceo-format-mode');
|
||||
expect(selectedBy('plan-ceo-review/SKILL.md.tmpl')).toContain('codex-plan-ceo-format-approach');
|
||||
expect(selectedBy('plan-eng-review/SKILL.md.tmpl')).toContain('codex-plan-eng-format-coverage');
|
||||
expect(selectedBy('plan-eng-review/SKILL.md.tmpl')).toContain('codex-plan-eng-format-kind');
|
||||
expect(selectedBy('test/codex-e2e-plan-format.test.ts').length).toBe(4);
|
||||
});
|
||||
|
||||
test('Sol fixture generation changes select its periodic case', () => {
|
||||
for (const file of ['test/helpers/sol-skill-fixture.ts', 'test/sol-skill-fixture.test.ts']) {
|
||||
expect(selectedBy(file)).toEqual(['codex-sol-scope-termination']);
|
||||
|
||||
@@ -1,103 +0,0 @@
|
||||
import { expect, test } from 'bun:test';
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
import * as predicates from './helpers/claude-pty-runner';
|
||||
import { E2E_TOUCHFILES } from './helpers/touchfiles-data';
|
||||
import fixture from './fixtures/conductor-prose-ao.json';
|
||||
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||
|
||||
const partial=fixture.publicDecisionTail;
|
||||
// This next frame is synthetic; the retained live attempt ended during A.
|
||||
const complete=partial+'\nB) Keep all four components and define cache invalidation before implementation.\nReply with A or B.';
|
||||
async function observe(frames:string[],verdict:'waiting'|'working',required?:boolean){
|
||||
const source=fs.readFileSync(path.join(import.meta.dir,'helpers/claude-pty-runner.ts'),'utf8');
|
||||
const start=source.indexOf('export async function runPlanSkillObservation('),end=source.indexOf('\n// ─',start);
|
||||
expect(start).toBeGreaterThan(0);expect(end).toBeGreaterThan(start);
|
||||
const js=new Bun.Transpiler({loader:'ts'}).transformSync(source.slice(start,end).replace('export async function','async function')+'\nreturn runPlanSkillObservation;');
|
||||
let clock=0,tick=-1,closed=0,judged=0;
|
||||
const current=()=>frames[Math.min(Math.max(tick,0),frames.length-1)]!;
|
||||
const args:Record<string,unknown>={path,process:{cwd:()=>'/synthetic-owned'},Date:{now:()=>clock},randomUUID:()=> 'owned',
|
||||
Bun:{sleep:async(ms:number)=>{if(ms===2000){tick++;clock+=61000;}else clock+=ms;}},
|
||||
launchClaudePty:async()=>({send:()=>{},mark:()=>0,exited:()=>false,visibleSince:current,rawOutput:current,currentScreen:async()=>current(),hermeticConfigDir:null,close:async()=>{closed++;}}),
|
||||
createPlanCountSnapshotWriter:()=>()=>({}),logPtySnapshot:()=>{},
|
||||
submitPlanSeed: async () => {}, PlanSeedTimeout: class extends Error {},
|
||||
isRejectedSlashCommand:predicates.isRejectedSlashCommand,
|
||||
isProseAUQVisible:predicates.isProseAUQVisible,isPlanReadyVisible:predicates.isPlanReadyVisible,
|
||||
isUnknownSlashCommandVisible:predicates.isUnknownSlashCommandVisible,
|
||||
isScopeGateQuestionVisible:predicates.isScopeGateQuestionVisible,isScopeGateAutoSelectVisible:predicates.isScopeGateAutoSelectVisible,
|
||||
classifyVisible:predicates.classifyVisible,extractPlanFilePath:predicates.extractPlanFilePath,findNativeAutoDecision:()=>null,
|
||||
judgePtyState:()=>{judged++;return {state:verdict,reasoning:'synthetic fixed verdict'};},
|
||||
};
|
||||
const run=new Function(...Object.keys(args),js)(...Object.values(args));
|
||||
const obs=await run({skillName:'plan-eng-review',initialPlanContent:'# Plan: Required draft',timeoutMs:300000,...(required===undefined?{}:{requireProseEvidence:required})});
|
||||
expect(closed).toBe(1);
|
||||
return {obs,polls:tick+1,judged};
|
||||
}
|
||||
|
||||
test('a judge waiting on the exact partial Conductor brief cannot stop a prose-required observation',async()=>{
|
||||
expect(fixture.actualFlags.proseAUQEverObserved).toBe(false);expect(fixture.actualFlags.waitingEverObserved).toBe(true);
|
||||
expect(predicates.isProseAUQVisible(partial)).toBe(false);expect(predicates.isProseAUQVisible(complete)).toBe(true);
|
||||
const {obs,polls,judged}=await observe([partial,complete],'waiting',true);
|
||||
expect(polls).toBe(2);expect(judged).toBe(1);expect(obs.outcome).toBe('asked');
|
||||
expect(obs.proseAUQEverObserved).toBe(true);expect(obs.waitingEverObserved).toBe(true);
|
||||
});
|
||||
test('partial-only judge waiting reaches the existing budget without gaining prose fallback credit',async()=>{
|
||||
const {obs,polls,judged}=await observe([partial],'waiting',true);
|
||||
expect(polls).toBe(5);expect(judged).toBe(5);expect(obs.outcome).toBe('timeout');
|
||||
expect(obs.proseAUQEverObserved).toBe(false);expect(obs.waitingEverObserved).toBe(true);
|
||||
});
|
||||
test('the prose requirement does not alter completed questions or deterministic failure precedence',async()=>{
|
||||
const done=await observe([complete],'waiting',true);
|
||||
expect(done.obs.outcome).toBe('asked');expect(done.obs.proseAUQEverObserved).toBe(true);expect(done.judged).toBe(0);
|
||||
const wrote=await observe(['⏺ Write(/tmp/foreign-output.md)'],'waiting',true);
|
||||
expect(wrote.obs.outcome).toBe('silent_write');expect(wrote.obs.proseAUQEverObserved).toBe(false);expect(wrote.judged).toBe(0);
|
||||
});
|
||||
test('other callers retain the original judge waiting behavior',async()=>{
|
||||
for(const required of [undefined,false]){
|
||||
const {obs,polls}=await observe([partial,complete],'waiting',required);
|
||||
expect(polls).toBe(1);expect(obs.outcome).toBe('asked');expect(obs.proseAUQEverObserved).toBe(false);expect(obs.waitingEverObserved).toBe(true);
|
||||
}
|
||||
const working=await observe([partial],'working',true);
|
||||
expect(working.obs.outcome).toBe('timeout');expect(working.obs.waitingEverObserved).toBe(false);
|
||||
});
|
||||
async function exerciseCaller(outcome: string, proseAUQEverObserved: boolean) {
|
||||
const caller=fs.readFileSync(path.join(import.meta.dir,'skill-e2e-conductor-prose.test.ts'),'utf8');
|
||||
const callbacks: Array<() => Promise<void>> = [];
|
||||
let calls = 0;
|
||||
const bindings = {
|
||||
expect, CAPTURE_MS, CAPTURE_LONG_MS,
|
||||
describeE2ETier: (tier: string) => {
|
||||
expect(tier).toBe('periodic');
|
||||
return (_title: string, register: () => void) => register();
|
||||
},
|
||||
test: (_title: string, callback: () => Promise<void>, timeout: number) => {
|
||||
expect(timeout).toBe(CAPTURE_LONG_MS); callbacks.push(callback);
|
||||
},
|
||||
runPlanSkillObservation: async (opts: Record<string, unknown>) => {
|
||||
calls++;
|
||||
expect(opts).toMatchObject({skillName: 'plan-eng-review', inPlanMode: true,
|
||||
requireProseEvidence: true, timeoutMs: CAPTURE_MS,
|
||||
env: {CONDUCTOR_WORKSPACE_PATH: '/tmp/conductor-prose-e2e'},
|
||||
extraArgs: ['--disallowedTools', 'AskUserQuestion']});
|
||||
return {outcome, proseAUQEverObserved, summary: 'controlled caller', evidence: partial};
|
||||
},
|
||||
};
|
||||
const body = caller.replace(/^import[\s\S]*?;\n/gm, '');
|
||||
new Function(...Object.keys(bindings), new Bun.Transpiler({loader: 'ts'}).transformSync(body))(...Object.values(bindings));
|
||||
expect(callbacks).toHaveLength(1);
|
||||
try { await callbacks[0]!(); }
|
||||
finally { expect(calls).toBe(1); }
|
||||
}
|
||||
test('the actual Conductor caller requests prose evidence and accepts a completed decision', async () => {
|
||||
await exerciseCaller('asked', true);
|
||||
});
|
||||
test.each(['asked', 'auto_decided', 'plan_ready'])('the actual Conductor caller rejects %s without independent prose evidence', async outcome => {
|
||||
await expect(exerciseCaller(outcome, false)).rejects.toThrow('Conductor prose decision not observed');
|
||||
});
|
||||
test.each(['silent_write', 'timeout', 'exited'])('the actual Conductor caller preserves the %s failure even with earlier prose evidence', async outcome => {
|
||||
await expect(exerciseCaller(outcome, true)).rejects.toThrow(outcome === 'silent_write'
|
||||
? 'skill wrote findings without surfacing a decision' : `outcome=${outcome}`);
|
||||
});
|
||||
test('the Conductor regression and public fixture select their existing owner',()=>{
|
||||
for(const p of ['test/conductor-prose-observation-ao.test.ts','test/fixtures/conductor-prose-ao.json'])expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(p)).map(([owner])=>owner)).toEqual(['conductor-prose']);
|
||||
});
|
||||
@@ -41,8 +41,8 @@ test('the existing quality and behavior phases retain their complete separate sh
|
||||
const behaviorFiles = behavior.entries.filter(entry => entry.status === 'planned').map(entry => entry.file);
|
||||
expect(quality.evalsAll).toBe(true);
|
||||
expect(behavior.evalsAll).toBe(true);
|
||||
expect(qualityFiles).toHaveLength(2);
|
||||
expect(behaviorFiles).toHaveLength(56);
|
||||
expect(qualityFiles).toHaveLength(1);
|
||||
expect(behaviorFiles).toHaveLength(51);
|
||||
expect(qualityFiles.every(file => file.startsWith('test/skill-llm-eval'))).toBe(true);
|
||||
expect(behaviorFiles.every(file => !qualityFiles.includes(file))).toBe(true);
|
||||
});
|
||||
|
||||
@@ -29,7 +29,7 @@ describe('AO completed manual DX handoff preserves report freshness',()=>{
|
||||
expect(E2E_TOUCHFILES[owner]).toContain('test/fixtures/dx-manual-handoff-ao.json');
|
||||
}
|
||||
const arrays=[...Object.values(E2E_TOUCHFILES),...Object.values(LLM_JUDGE_TOUCHFILES),GLOBAL_TOUCHFILES];
|
||||
expect(arrays).toHaveLength(244);
|
||||
expect(arrays).toHaveLength(221);
|
||||
for(const values of arrays)for(let i=0;i<values.length;i++)expect(typeof values[i]).toBe('string');
|
||||
});
|
||||
test('exact owned report precedes navigation only, with the current Exit gate recognized',()=>{
|
||||
|
||||
@@ -57,8 +57,6 @@ const KNOWN_UNREGISTERED = new Set([
|
||||
'test/skill-e2e-auq-matrix.test.ts',
|
||||
// Standalone periodic self-gated A/B probe; template-literal testNames (auq-ab-${label}), no E2E map key — fail-open-safe, runs on every periodic sweep.
|
||||
'test/skill-e2e-auq-verbose-vs-carved-ab.test.ts',
|
||||
// bin-script pipeline test (spawns bun scripts, no model spend) that lives under the skill-e2e-* glob; no E2E map key exists for it.
|
||||
'test/skill-e2e-memory-pipeline.test.ts',
|
||||
]);
|
||||
|
||||
describe('E2E tier alignment (touchfiles declaration vs test self-gate)', () => {
|
||||
|
||||
@@ -127,16 +127,16 @@ test('live periodic census fits the declared CI wall including setup', () => {
|
||||
return paidShardWallUpperBoundMs(files, workers);
|
||||
});
|
||||
expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(periodicJob['timeout-minutes'] * 60_000);
|
||||
expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(100);
|
||||
expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(82);
|
||||
const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount - 1);
|
||||
expect(overlays).toHaveLength(6);
|
||||
expect(overlays).toHaveLength(4);
|
||||
expect(overlays.every(e => isOverlayTestFile(e.file))).toBe(true);
|
||||
expect(m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount).map(e => e.file)).toEqual([AUTOPLAN_CHAIN_BUDGET.file]);
|
||||
});
|
||||
|
||||
test('registered allocation is deterministic and preserves every discovered file', () => {
|
||||
const files = collectPaidTestFiles();
|
||||
expect(files).toHaveLength(119);
|
||||
expect(files).toHaveLength(105);
|
||||
const m = livePlan(files);
|
||||
expect(livePlan([...files].reverse())).toEqual(m);
|
||||
expect(m.entries.map(e => e.file).sort()).toEqual([...files].sort());
|
||||
@@ -175,7 +175,7 @@ test('current detach supervision covers the live-census floor', () => {
|
||||
const floor = Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05);
|
||||
const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8'));
|
||||
const configured = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]);
|
||||
expect(floor).toBe(65541);
|
||||
expect(floor).toBe(57330);
|
||||
expect(configured).toBeGreaterThanOrEqual(floor);
|
||||
expect(pkg.scripts['eval:bg:gate']).toContain('--timeout 36000');
|
||||
});
|
||||
|
||||
@@ -184,7 +184,7 @@ test('rejected completion does not erase a genuine earlier question or change un
|
||||
|
||||
test('completion evidence dependencies select exactly the seeded observation owners', () => {
|
||||
const owners = ['plan-ceo-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-design-review-plan-mode',
|
||||
'plan-devex-review-plan-mode', 'plan-mode-no-op', 'auto-decide-preserved', 'conductor-prose'].sort();
|
||||
'plan-devex-review-plan-mode', 'plan-mode-no-op', 'auto-decide-preserved'].sort();
|
||||
for (const file of ['test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json']) {
|
||||
expect(selectTests([file], E2E_TOUCHFILES).selected.sort()).toEqual(owners);
|
||||
}
|
||||
|
||||
-24
@@ -1,24 +0,0 @@
|
||||
{
|
||||
"publicDecisionTail": "D1—ReducethepricingtiertoStripe-nativepieces,orbuildallfourcomponents?\r\rReplywithaletter(Conductorsession,sothisbriefisinprose).Project/branch:gstackedinburgh-v1,reviewingan\rexternal pricing-tier plan.\r\rELI10:Youwantmoredeveloperstosignup,soyou'readdingacheaperplan.Thedraftbuildsfournewthingstodo\rit: a Stripe price, a custm pricing web age,r own atabse tbletht remmbers who paid fowhat, ad a fast\rn-memory copy of tht table. Strpealready ships the table (Entitlements)and th page (rcing table). Every exra\rcopy of \"wh is allowed to dowhat\" is another place that can disagre withStripe when a card fails or awebhook\rarriveslate. The stakes: a developer who paidand sees \"upgrae required,\"or ono cnelled andkeeps access.\r\rStakesifwepickwrong:threesourcesoftruthforentitlementswithnoinvalidationdesignmeanssilentaccessbugs\rtht onlyshow upas angry support tickets.\r\rRecommendation:Abecauseitkeepseveryuser-visibleoutcomeofthedraftwhiledeletingthetwocomponentsmost\rlikely to causeentitlemnt drift,and it fits \"engineered enough\" an \"smallestiff that cleanly exressethe\rchange.\"\r\rCompleteness:A=9/10,B=7/10.Cisahold,notabuildoption,sonoscore.\r\rA)Stripe-nativeslimbuild(recommended).Completeness9/10.Human:~1week/CC:~1to2hours.AddoneStripePrice\rplus a Fature-to-Product mapping, mirror active_entitlement_sumryinto a singleexisting cstomer table hrough an\ridempotent webhok, gate fetures off that olumn, ship the pricing surfacebehinda feature flag (Stripe pricin\u001b[?25h",
|
||||
"actualFlags": {
|
||||
"proseAUQEverObserved": false,
|
||||
"waitingEverObserved": true,
|
||||
"scopeGateAutoSelectObserved": true
|
||||
},
|
||||
"owner": {
|
||||
"sessionId": "4e820ffc-b6dd-47d7-b074-5c4b0a3cbb68",
|
||||
"commandStartedAt": 1789041170391,
|
||||
"nativePolledAt": 1789041383968,
|
||||
"captureAt": "2026-09-10T11:56:33.483Z"
|
||||
},
|
||||
"provenance": {
|
||||
"observation": ".context/ship-source-ao-delta-paid-20260910-v1/conductor-first-failure-ledger-v1/observation.json",
|
||||
"observationSha256": "da8c8a342f44ae06bfdd092f1bdb6a395765553d65ac99eecaaa35dedb34545d",
|
||||
"visible": ".context/ship-source-ao-delta-paid-20260910-v1/conductor-first-failure-ledger-v1/terminal.visible.log",
|
||||
"visibleSha256": "2a2fca3e8d1b51694d76a3dbcfac524626f856e4ebf38f4bd093346edb64bf82",
|
||||
"startCharacterOffset": 78469,
|
||||
"publicTailSha256": "e551890b1d839cee7d7dce05120b8dc05a54f27fe8f69617aa99384c49316604",
|
||||
"limit": "Exact public partial D1 tail; no completed native prose message retained. A later second option in tests is synthetic.",
|
||||
"offsetUnit": "Unicode code points in decoded retained bytes, without newline normalization"
|
||||
}
|
||||
}
|
||||
Vendored
-2
@@ -1,6 +1,4 @@
|
||||
{
|
||||
"command_reference": { "clarity": 4, "completeness": 3, "actionability": 4 },
|
||||
"snapshot_flags": { "clarity": 4, "completeness": 4, "actionability": 4 },
|
||||
"browse_skill": { "clarity": 4, "completeness": 4, "actionability": 4 },
|
||||
"qa_workflow": { "clarity": 4, "completeness": 4, "actionability": 4 },
|
||||
"qa_health_rubric": { "clarity": 4, "completeness": 3, "actionability": 4 }
|
||||
|
||||
Vendored
-44
@@ -305,50 +305,6 @@ export const OVERLAY_FIXTURES: OverlayFixture[] = [
|
||||
pass: lowerIsBetter20Pct,
|
||||
},
|
||||
|
||||
{
|
||||
id: 'opus-4-7-effort-match-trivial-sonnet',
|
||||
overlayPath: 'model-overlays/opus-4-7.md',
|
||||
model: 'claude-sonnet-4-6',
|
||||
trials: 10,
|
||||
concurrency: 3,
|
||||
direction: 'lower_is_better',
|
||||
maxTurns: 8,
|
||||
setupWorkspace: (dir) => {
|
||||
fs.writeFileSync(
|
||||
path.join(dir, 'config.json'),
|
||||
'{"name": "demo", "version": "1.0.0"}\n',
|
||||
);
|
||||
},
|
||||
userPrompt: "What's the version in config.json? Return only a JSON object with the version key and its exact string value. " +
|
||||
"The final message is consumed directly by JSON.parse: return the exact JSON object, with no Markdown fences and no other prose.",
|
||||
metric: reportedThinkingTokens,
|
||||
metricName: 'reported_thinking_tokens',
|
||||
verify: (r) => assertFinalJson(r, { version: '1.0.0' }),
|
||||
comparison: { direction: 'lower_is_better', minimum: 0 },
|
||||
pass: lowerIsBetter20Pct,
|
||||
},
|
||||
|
||||
{
|
||||
id: 'opus-4-7-literal-interpretation-sonnet',
|
||||
overlayPath: 'model-overlays/opus-4-7.md',
|
||||
model: 'claude-sonnet-4-6',
|
||||
trials: 10,
|
||||
concurrency: 3,
|
||||
direction: 'higher_is_better',
|
||||
allowedTools: ['Read', 'Glob', 'Grep', 'Bash', 'Edit', 'Write'],
|
||||
maxTurns: 15,
|
||||
setupWorkspace: setupLiteralWorkspace,
|
||||
userPrompt: 'Fix the failing tests. Preserve the specified behavior and repair the implementation.',
|
||||
metric: (_r, dir, deadlineAt) => {
|
||||
if (!dir) throw new Error('literal fixture metric needs its workspace');
|
||||
return correctLiteralTargets(dir, deadlineAt);
|
||||
},
|
||||
metricName: 'correct_target_behaviors',
|
||||
taskCorrect: (metric) => metric === 3,
|
||||
allowedChanges: ['src/auth.ts', 'src/billing.ts', 'src/notifications.ts'],
|
||||
comparison: { direction: 'higher_is_better', minimum: 0, maximum: 3 },
|
||||
pass: higherIsBetter20Pct,
|
||||
},
|
||||
];
|
||||
|
||||
// Validate at module load so a broken fixture fails fast at test startup,
|
||||
|
||||
@@ -1,174 +0,0 @@
|
||||
/**
|
||||
* Gemini CLI E2E smoke test — verify Gemini CLI can start and discover skills.
|
||||
*
|
||||
* This is a lightweight smoke test, not a full integration test. Gemini CLI
|
||||
* gets lost in worktrees and times out on complex tasks. The smoke test
|
||||
* validates that the skill files are structured correctly for Gemini's
|
||||
* .agents/skills/ discovery mechanism.
|
||||
*
|
||||
* Prerequisites:
|
||||
* - `gemini` binary installed (npm install -g @google/gemini-cli)
|
||||
* - Gemini authenticated via ~/.gemini/ config or GEMINI_API_KEY env var
|
||||
* - EVALS=1 env var set (same gate as Claude E2E tests)
|
||||
*
|
||||
* Skips gracefully when prerequisites are not met.
|
||||
*/
|
||||
|
||||
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
|
||||
import { JUDGE_MS } from './helpers/eval-budgets';
|
||||
import { runGeminiSkill } from './helpers/gemini-session-runner';
|
||||
import type { GeminiResult } from './helpers/gemini-session-runner';
|
||||
import { EvalCollector } from './helpers/eval-store';
|
||||
import { selectTests, detectBaseBranch, getChangedFiles, E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from './helpers/touchfiles';
|
||||
import { createTestWorktree, harvestAndCleanup } from './helpers/e2e-helpers';
|
||||
import * as path from 'path';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
|
||||
// --- Prerequisites check ---
|
||||
|
||||
const GEMINI_AVAILABLE = (() => {
|
||||
try {
|
||||
const result = Bun.spawnSync(['which', 'gemini'], { timeout: 30_000 });
|
||||
return result.exitCode === 0;
|
||||
} catch { return false; }
|
||||
})();
|
||||
|
||||
// A binary on PATH is not enough: the CLI can be present but UNUSABLE — the
|
||||
// individual code-assist auth path was deprecated upstream ("migrate to the
|
||||
// Antigravity suite"), which fails every run before any model call, and flag
|
||||
// churn (--skip-trust removed in 0.34) errors at argv parse. Probe with a
|
||||
// bare --help: a CLI that can't even print usage is unusable, and a working
|
||||
// one is cheap to confirm. Deeper auth failures are classified per-run.
|
||||
const GEMINI_USABLE = GEMINI_AVAILABLE && (() => {
|
||||
try {
|
||||
const result = Bun.spawnSync(['gemini', '--help'], { timeout: 15_000 });
|
||||
return result.exitCode === 0;
|
||||
} catch { return false; }
|
||||
})();
|
||||
|
||||
const evalsEnabled = !!process.env.EVALS;
|
||||
|
||||
// External-service tests are periodic-tier (CLAUDE.md tiering rule 3):
|
||||
// "Requires external service (Codex, Gemini)? -> periodic". The positive
|
||||
// form below is the canonical whole-file guard shape — the sharded runner's
|
||||
// classifyPaidTestFile greps for it to exclude this file from gate.
|
||||
const tierOk = process.env.EVALS_TIER === 'periodic';
|
||||
|
||||
// Skip all tests if gemini is not available/usable, EVALS is not set, or
|
||||
// we're in the gate tier.
|
||||
const SKIP = !GEMINI_USABLE || !evalsEnabled || !tierOk;
|
||||
|
||||
const describeGemini = SKIP ? describe.skip : describe;
|
||||
|
||||
// Log why we're skipping (helpful for debugging CI)
|
||||
if (!evalsEnabled) {
|
||||
// Silent — same as Claude E2E tests, EVALS=1 required
|
||||
} else if (!tierOk) {
|
||||
process.stderr.write('\nGemini E2E: SKIPPED — external-service test, periodic tier only (EVALS_TIER === \'periodic\')\n');
|
||||
} else if (!GEMINI_AVAILABLE) {
|
||||
process.stderr.write('\nGemini E2E: SKIPPED — gemini binary not found (install: npm i -g @google/gemini-cli)\n');
|
||||
} else if (!GEMINI_USABLE) {
|
||||
process.stderr.write('\nGemini E2E: SKIPPED — gemini CLI present but unusable (auth path deprecated upstream or CLI broken; try updating @google/gemini-cli)\n');
|
||||
}
|
||||
|
||||
// --- Diff-based test selection ---
|
||||
|
||||
// Gemini E2E touchfiles — DERIVED from the canonical map, never a local fork
|
||||
// (the old hand-copy kept a gitignored '.agents/skills/**' pattern that can
|
||||
// never match a git diff and missed canonical deps — same drift class as the
|
||||
// codex copy).
|
||||
const GEMINI_E2E_TOUCHFILES: Record<string, string[]> = Object.fromEntries(
|
||||
(['gemini-smoke'] as const).map((key) => {
|
||||
if (!E2E_TOUCHFILES[key]) throw new Error(`canonical E2E_TOUCHFILES lost key '${key}' — fix the map, not this file`);
|
||||
return [key, E2E_TOUCHFILES[key]];
|
||||
}),
|
||||
);
|
||||
|
||||
let selectedTests: string[] | null = null; // null = run all
|
||||
|
||||
if (evalsEnabled && !process.env.EVALS_ALL) {
|
||||
const baseBranch = process.env.EVALS_BASE
|
||||
|| detectBaseBranch(ROOT)
|
||||
|| 'main';
|
||||
const changedFiles = getChangedFiles(baseBranch, ROOT);
|
||||
|
||||
if (changedFiles.length > 0) {
|
||||
const selection = selectTests(changedFiles, GEMINI_E2E_TOUCHFILES, GLOBAL_TOUCHFILES);
|
||||
selectedTests = selection.selected;
|
||||
process.stderr.write(`\nGemini E2E selection (${selection.reason}): ${selection.selected.length}/${Object.keys(GEMINI_E2E_TOUCHFILES).length} tests\n`);
|
||||
if (selection.skipped.length > 0) {
|
||||
process.stderr.write(` Skipped: ${selection.skipped.join(', ')}\n`);
|
||||
}
|
||||
process.stderr.write('\n');
|
||||
}
|
||||
}
|
||||
|
||||
/** Skip an individual test if not selected by diff-based selection. */
|
||||
function testIfSelected(testName: string, fn: () => Promise<void>, timeout: number) {
|
||||
const shouldRun = selectedTests === null || selectedTests.includes(testName);
|
||||
(shouldRun ? test.concurrent : test.skip)(testName, fn, timeout);
|
||||
}
|
||||
|
||||
// --- Eval result collector ---
|
||||
|
||||
const evalCollector = evalsEnabled && !SKIP ? new EvalCollector('e2e-gemini') : null;
|
||||
|
||||
function recordGeminiE2E(name: string, result: GeminiResult, passed: boolean) {
|
||||
evalCollector?.addTest({
|
||||
name,
|
||||
suite: 'gemini-e2e',
|
||||
tier: 'e2e',
|
||||
passed,
|
||||
duration_ms: result.durationMs,
|
||||
cost_usd: 0,
|
||||
output: result.output?.slice(0, 2000),
|
||||
turns_used: result.toolCalls.length,
|
||||
exit_reason: result.exitCode === 0 ? 'success' : `exit_code_${result.exitCode}`,
|
||||
});
|
||||
}
|
||||
|
||||
function logGeminiCost(label: string, result: GeminiResult) {
|
||||
const durationSec = Math.round(result.durationMs / 1000);
|
||||
console.log(`${label}: ${result.tokens} tokens, ${result.toolCalls.length} tool calls, ${durationSec}s`);
|
||||
}
|
||||
|
||||
// Finalize eval results on exit
|
||||
afterAll(async () => {
|
||||
if (evalCollector) {
|
||||
await evalCollector.finalize();
|
||||
}
|
||||
});
|
||||
|
||||
// --- Tests ---
|
||||
|
||||
describeGemini('Gemini E2E', () => {
|
||||
let testWorktree: string;
|
||||
|
||||
beforeAll(() => {
|
||||
testWorktree = createTestWorktree('gemini');
|
||||
});
|
||||
|
||||
afterAll(() => {
|
||||
harvestAndCleanup('gemini');
|
||||
});
|
||||
|
||||
testIfSelected('gemini-smoke', async () => {
|
||||
// Smoke test: can Gemini start, read the repo, and produce output?
|
||||
// Uses a simple prompt that doesn't require skill invocation or complex navigation.
|
||||
const result = await runGeminiSkill({
|
||||
prompt: 'What is this project? Answer in one sentence based on the README.',
|
||||
timeoutMs: JUDGE_MS,
|
||||
cwd: testWorktree,
|
||||
});
|
||||
|
||||
logGeminiCost('gemini-smoke', result);
|
||||
|
||||
// Pass if Gemini produced any meaningful output (even with non-zero exit from timeout)
|
||||
const hasOutput = result.output.length > 10;
|
||||
const passed = hasOutput;
|
||||
recordGeminiE2E('gemini-smoke', result, passed);
|
||||
|
||||
expect(result.output.length, 'Gemini should produce output').toBeGreaterThan(10);
|
||||
}, JUDGE_MS);
|
||||
});
|
||||
@@ -369,6 +369,40 @@ describe('gstack-skill-start behavior', () => {
|
||||
const out = runStart();
|
||||
expect(out).toMatch(/^ARTIFACTS_SYNC: off$/m);
|
||||
});
|
||||
|
||||
test('artifacts-sync consent is asked before any artifacts egress, and only in interactive sessions', () => {
|
||||
const gh = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ss-privacy-'));
|
||||
const bin = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ss-privacy-bin-'));
|
||||
const remote = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ss-privacy-remote-'));
|
||||
const git = (args: string[], cwd: string) => execFileSync('git', args, { cwd, timeout: 30_000, stdio: 'pipe' });
|
||||
try {
|
||||
fs.writeFileSync(path.join(bin, 'gbrain'), '#!/bin/sh\nexit 0\n', { mode: 0o755 });
|
||||
git(['init', '-q', '--bare'], remote);
|
||||
git(['init', '-q', '-b', 'main'], gh);
|
||||
git(['remote', 'add', 'origin', remote], gh);
|
||||
const pullStamp = path.join(gh, '.brain-last-pull');
|
||||
const start = (config: string, env: Record<string, string> = {}) => {
|
||||
fs.writeFileSync(path.join(gh, 'config.yaml'), `update_check: false\n${config}`);
|
||||
return runStart([], { GSTACK_HOME: gh, PATH: `${bin}${path.delimiter}${process.env.PATH}`, ...env });
|
||||
};
|
||||
const gates = (out: string) => (out.match(/^GSTACK_INSTRUCTION_BEGIN: privacy-stop-gate/gm) ?? []).length;
|
||||
|
||||
const pending = start('');
|
||||
expect(gates(pending)).toBe(1);
|
||||
expect(pending).toContain('How much should sync?');
|
||||
expect(pending).toMatch(/^ARTIFACTS_SYNC: off$/m);
|
||||
expect(fs.existsSync(pullStamp)).toBe(false);
|
||||
|
||||
expect(gates(start('', { GSTACK_SESSION_KIND: 'spawned' }))).toBe(0);
|
||||
expect(fs.existsSync(pullStamp)).toBe(false);
|
||||
|
||||
const consented = start('artifacts_sync_mode: full\nartifacts_sync_mode_prompted: true\n');
|
||||
expect(gates(consented)).toBe(0);
|
||||
expect(fs.existsSync(pullStamp)).toBe(true);
|
||||
} finally {
|
||||
for (const dir of [gh, bin, remote]) fs.rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('gstack-skill-end', () => {
|
||||
|
||||
@@ -106,9 +106,8 @@ export const FILE_RETRY_BUDGETS = [
|
||||
...STRICT_RETRY_CASE_BUDGETS,
|
||||
...[
|
||||
// Sixteen workflow judges include their 10s recording grace; the other
|
||||
// eleven judges retain 120s. Supervise all 27 and the existing one retry.
|
||||
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 16 * (JUDGE_MS + 10_000) + 11 * JUDGE_MS, retries: 1 },
|
||||
{ file: 'test/codex-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_LONG_MS + 10_000), retries: 1 },
|
||||
// seven judges retain 120s. Supervise all 23 and the existing one retry.
|
||||
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 16 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 },
|
||||
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
|
||||
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, retries: 1 },
|
||||
|
||||
@@ -1,104 +0,0 @@
|
||||
import { describe, test, expect } from 'bun:test';
|
||||
import { parseGeminiJSONL } from './gemini-session-runner';
|
||||
|
||||
// Fixture: actual Gemini CLI stream-json output with tool use
|
||||
const FIXTURE_LINES = [
|
||||
'{"type":"init","timestamp":"2026-03-20T15:14:46.455Z","session_id":"test-session-123","model":"auto-gemini-3"}',
|
||||
'{"type":"message","timestamp":"2026-03-20T15:14:46.456Z","role":"user","content":"list the files"}',
|
||||
'{"type":"message","timestamp":"2026-03-20T15:14:49.650Z","role":"assistant","content":"I will list the files.","delta":true}',
|
||||
'{"type":"tool_use","timestamp":"2026-03-20T15:14:49.690Z","tool_name":"run_shell_command","tool_id":"cmd_1","parameters":{"command":"ls"}}',
|
||||
'{"type":"tool_result","timestamp":"2026-03-20T15:14:49.931Z","tool_id":"cmd_1","status":"success","output":"file1.ts\\nfile2.ts"}',
|
||||
'{"type":"message","timestamp":"2026-03-20T15:14:51.945Z","role":"assistant","content":"Here are the files.","delta":true}',
|
||||
'{"type":"result","timestamp":"2026-03-20T15:14:52.030Z","status":"success","stats":{"total_tokens":27147,"input_tokens":26928,"output_tokens":87,"cached":0,"duration_ms":5575,"tool_calls":1}}',
|
||||
];
|
||||
|
||||
describe('parseGeminiJSONL', () => {
|
||||
test('extracts session ID from init event', () => {
|
||||
const parsed = parseGeminiJSONL(FIXTURE_LINES);
|
||||
expect(parsed.sessionId).toBe('test-session-123');
|
||||
});
|
||||
|
||||
test('concatenates assistant message deltas into output', () => {
|
||||
const parsed = parseGeminiJSONL(FIXTURE_LINES);
|
||||
expect(parsed.output).toBe('I will list the files.Here are the files.');
|
||||
});
|
||||
|
||||
test('ignores user messages', () => {
|
||||
const lines = [
|
||||
'{"type":"message","role":"user","content":"this should be ignored"}',
|
||||
'{"type":"message","role":"assistant","content":"this should be kept","delta":true}',
|
||||
];
|
||||
const parsed = parseGeminiJSONL(lines);
|
||||
expect(parsed.output).toBe('this should be kept');
|
||||
});
|
||||
|
||||
test('extracts tool names from tool_use events', () => {
|
||||
const parsed = parseGeminiJSONL(FIXTURE_LINES);
|
||||
expect(parsed.toolCalls).toHaveLength(1);
|
||||
expect(parsed.toolCalls[0]).toBe('run_shell_command');
|
||||
});
|
||||
|
||||
test('extracts total tokens from result stats', () => {
|
||||
const parsed = parseGeminiJSONL(FIXTURE_LINES);
|
||||
expect(parsed.tokens).toBe(27147);
|
||||
});
|
||||
|
||||
test('skips malformed lines without throwing', () => {
|
||||
const lines = [
|
||||
'{"type":"init","session_id":"ok"}',
|
||||
'this is not json',
|
||||
'{"type":"message","role":"assistant","content":"hello","delta":true}',
|
||||
'{incomplete json',
|
||||
'{"type":"result","status":"success","stats":{"total_tokens":100}}',
|
||||
];
|
||||
const parsed = parseGeminiJSONL(lines);
|
||||
expect(parsed.sessionId).toBe('ok');
|
||||
expect(parsed.output).toBe('hello');
|
||||
expect(parsed.tokens).toBe(100);
|
||||
});
|
||||
|
||||
test('skips empty and whitespace-only lines', () => {
|
||||
const lines = [
|
||||
'',
|
||||
' ',
|
||||
'{"type":"init","session_id":"s1"}',
|
||||
'\t',
|
||||
'{"type":"result","status":"success","stats":{"total_tokens":50}}',
|
||||
];
|
||||
const parsed = parseGeminiJSONL(lines);
|
||||
expect(parsed.sessionId).toBe('s1');
|
||||
expect(parsed.tokens).toBe(50);
|
||||
});
|
||||
|
||||
test('handles empty input', () => {
|
||||
const parsed = parseGeminiJSONL([]);
|
||||
expect(parsed.output).toBe('');
|
||||
expect(parsed.toolCalls).toHaveLength(0);
|
||||
expect(parsed.tokens).toBe(0);
|
||||
expect(parsed.sessionId).toBeNull();
|
||||
});
|
||||
|
||||
test('handles missing fields gracefully', () => {
|
||||
const lines = [
|
||||
'{"type":"init"}', // no session_id
|
||||
'{"type":"message","role":"assistant"}', // no content
|
||||
'{"type":"tool_use"}', // no tool_name
|
||||
'{"type":"result","status":"success"}', // no stats
|
||||
];
|
||||
const parsed = parseGeminiJSONL(lines);
|
||||
expect(parsed.sessionId).toBeNull();
|
||||
expect(parsed.output).toBe('');
|
||||
expect(parsed.toolCalls).toHaveLength(0);
|
||||
expect(parsed.tokens).toBe(0);
|
||||
});
|
||||
|
||||
test('handles multiple tool_use events', () => {
|
||||
const lines = [
|
||||
'{"type":"tool_use","tool_name":"run_shell_command","tool_id":"cmd_1","parameters":{"command":"ls"}}',
|
||||
'{"type":"tool_use","tool_name":"read_file","tool_id":"cmd_2","parameters":{"path":"foo.ts"}}',
|
||||
'{"type":"tool_use","tool_name":"run_shell_command","tool_id":"cmd_3","parameters":{"command":"cat bar.ts"}}',
|
||||
];
|
||||
const parsed = parseGeminiJSONL(lines);
|
||||
expect(parsed.toolCalls).toEqual(['run_shell_command', 'read_file', 'run_shell_command']);
|
||||
});
|
||||
});
|
||||
@@ -1,262 +0,0 @@
|
||||
/**
|
||||
* Gemini CLI subprocess runner for skill E2E testing.
|
||||
*
|
||||
* Spawns `gemini -p` as an independent process, parses its stream-json
|
||||
* output, and returns structured results. Follows the same pattern as
|
||||
* codex-session-runner.ts but adapted for the Gemini CLI.
|
||||
*
|
||||
* Key differences from Codex session-runner:
|
||||
* - Uses `gemini -p` instead of `codex exec`
|
||||
* - Output is NDJSON with event types: init, message, tool_use, tool_result, result
|
||||
* - Uses `--output-format stream-json --yolo` instead of `--json -s read-only`
|
||||
* (`--skip-trust` was removed in gemini-cli 0.34; folder trust is settings-driven now)
|
||||
* - No temp HOME needed — Gemini discovers skills from `.agents/skills/` in cwd
|
||||
* - Message events are streamed with `delta: true` — must concatenate
|
||||
*/
|
||||
|
||||
import * as path from 'path';
|
||||
import { spawn } from 'child_process';
|
||||
import { Readable } from 'node:stream';
|
||||
import { hermeticChildEnv } from './hermetic-env';
|
||||
import { killProcessGroup } from '../../scripts/test-strict-output';
|
||||
|
||||
// --- Interfaces ---
|
||||
|
||||
export interface GeminiResult {
|
||||
output: string; // Full assistant message text (concatenated deltas)
|
||||
toolCalls: string[]; // Tool names from tool_use events
|
||||
tokens: number; // Total tokens used
|
||||
exitCode: number; // Process exit code
|
||||
durationMs: number; // Wall clock time
|
||||
sessionId: string | null; // Session ID from init event
|
||||
rawLines: string[]; // Raw JSONL lines for debugging
|
||||
}
|
||||
|
||||
// --- JSONL parser ---
|
||||
|
||||
export interface ParsedGeminiJSONL {
|
||||
output: string;
|
||||
toolCalls: string[];
|
||||
tokens: number;
|
||||
sessionId: string | null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse an array of JSONL lines from `gemini -p --output-format stream-json`.
|
||||
* Pure function — no I/O, no side effects.
|
||||
*
|
||||
* Handles these Gemini event types:
|
||||
* - init → extract session_id
|
||||
* - message (role=assistant, delta=true) → concatenate content into output
|
||||
* - tool_use → extract tool_name
|
||||
* - tool_result → logged but not extracted
|
||||
* - result → extract token usage from stats
|
||||
*/
|
||||
export function parseGeminiJSONL(lines: string[]): ParsedGeminiJSONL {
|
||||
const outputParts: string[] = [];
|
||||
const toolCalls: string[] = [];
|
||||
let tokens = 0;
|
||||
let sessionId: string | null = null;
|
||||
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue;
|
||||
try {
|
||||
const obj = JSON.parse(line);
|
||||
const t = obj.type || '';
|
||||
|
||||
if (t === 'init') {
|
||||
const sid = obj.session_id || '';
|
||||
if (sid) sessionId = sid;
|
||||
} else if (t === 'message') {
|
||||
if (obj.role === 'assistant' && obj.content) {
|
||||
outputParts.push(obj.content);
|
||||
}
|
||||
} else if (t === 'tool_use') {
|
||||
const name = obj.tool_name || '';
|
||||
if (name) toolCalls.push(name);
|
||||
} else if (t === 'result') {
|
||||
const stats = obj.stats || {};
|
||||
tokens = (stats.total_tokens || 0);
|
||||
}
|
||||
} catch { /* skip malformed lines */ }
|
||||
}
|
||||
|
||||
return {
|
||||
output: outputParts.join(''),
|
||||
toolCalls,
|
||||
tokens,
|
||||
sessionId,
|
||||
};
|
||||
}
|
||||
|
||||
// --- Main runner ---
|
||||
|
||||
/**
|
||||
* Run a prompt via `gemini -p` and return structured results.
|
||||
*
|
||||
* Spawns gemini with stream-json output, parses JSONL events,
|
||||
* and returns a GeminiResult. Skips gracefully if gemini binary is not found.
|
||||
*/
|
||||
export async function runGeminiSkill(opts: {
|
||||
prompt: string; // What to ask Gemini
|
||||
timeoutMs?: number; // Default 300000 (5 min)
|
||||
cwd?: string; // Working directory (where .agents/skills/ lives)
|
||||
}): Promise<GeminiResult> {
|
||||
const {
|
||||
prompt,
|
||||
timeoutMs = 300_000,
|
||||
cwd,
|
||||
} = opts;
|
||||
|
||||
const startTime = Date.now();
|
||||
|
||||
// Check if gemini binary exists
|
||||
const whichResult = Bun.spawnSync(['which', 'gemini'], { timeout: 30_000 });
|
||||
if (whichResult.exitCode !== 0) {
|
||||
return {
|
||||
output: 'SKIP: gemini binary not found',
|
||||
toolCalls: [],
|
||||
tokens: 0,
|
||||
exitCode: -1,
|
||||
durationMs: Date.now() - startTime,
|
||||
sessionId: null,
|
||||
rawLines: [],
|
||||
};
|
||||
}
|
||||
|
||||
// Build gemini command.
|
||||
// --skip-trust was REMOVED in gemini-cli 0.34 ("Unknown arguments:
|
||||
// skip-trust"); folder trust moved to settings and no longer needs a flag
|
||||
// for headless runs. --yolo still auto-approves tool actions.
|
||||
const args = ['-p', prompt, '--output-format', 'stream-json', '--yolo'];
|
||||
|
||||
// Spawn gemini — uses real HOME for auth (~/.gemini; HOME is allowlisted),
|
||||
// cwd for skill discovery. Hermetic scrub with gemini's auth surface
|
||||
// re-admitted (previously this spawn inherited the full operator env).
|
||||
// node:child_process spawn with `detached` (own process group) — mirrors
|
||||
// session-runner.ts. A bare kill signalled only gemini itself; tool
|
||||
// subprocesses survived as orphans holding our pipes open (the same
|
||||
// blocked-drain hang the claude runner fixed — this copy lacked it).
|
||||
const proc = spawn('gemini', args, {
|
||||
cwd: cwd || process.cwd(),
|
||||
stdio: ['ignore', 'pipe', 'pipe'],
|
||||
detached: process.platform !== 'win32',
|
||||
env: hermeticChildEnv(undefined, {
|
||||
extraAllow: ['GEMINI_API_KEY', 'GOOGLE_API_KEY', 'GOOGLE_APPLICATION_CREDENTIALS', 'GOOGLE_CLOUD_*', 'GEMINI_*'],
|
||||
}),
|
||||
});
|
||||
const stdoutWeb = Readable.toWeb(proc.stdout!) as ReadableStream<Uint8Array>;
|
||||
const stderrWeb = Readable.toWeb(proc.stderr!) as ReadableStream<Uint8Array>;
|
||||
const procExited: Promise<number> = new Promise((resolve) => {
|
||||
proc.on('close', (code) => resolve(code ?? 1));
|
||||
proc.on('error', () => resolve(1));
|
||||
});
|
||||
|
||||
// Race against timeout
|
||||
let timedOut = false;
|
||||
const timeoutId = setTimeout(() => {
|
||||
timedOut = true;
|
||||
// Group SIGKILL + reader cancel: kill the whole tree AND unblock the
|
||||
// read loop even if a stray grandchild survives the group kill.
|
||||
killProcessGroup(proc, 'SIGKILL');
|
||||
reader.cancel().catch(() => { /* stream already closed */ });
|
||||
}, timeoutMs);
|
||||
|
||||
// Stream and collect JSONL from stdout
|
||||
const collectedLines: string[] = [];
|
||||
const stderrPromise = new Response(stderrWeb).text();
|
||||
|
||||
const reader = stdoutWeb.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buf = '';
|
||||
|
||||
try {
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
buf += decoder.decode(value, { stream: true });
|
||||
const lines = buf.split('\n');
|
||||
buf = lines.pop() || '';
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue;
|
||||
collectedLines.push(line);
|
||||
|
||||
// Real-time progress to stderr
|
||||
try {
|
||||
const event = JSON.parse(line);
|
||||
if (event.type === 'tool_use' && event.tool_name) {
|
||||
const elapsed = Math.round((Date.now() - startTime) / 1000);
|
||||
process.stderr.write(` [gemini ${elapsed}s] tool: ${event.tool_name}\n`);
|
||||
} else if (event.type === 'message' && event.role === 'assistant' && event.content) {
|
||||
const elapsed = Math.round((Date.now() - startTime) / 1000);
|
||||
process.stderr.write(` [gemini ${elapsed}s] message: ${event.content.slice(0, 100)}\n`);
|
||||
}
|
||||
} catch { /* skip — parseGeminiJSONL will handle it later */ }
|
||||
}
|
||||
}
|
||||
} catch { /* stream read error — fall through to exit code handling */ }
|
||||
|
||||
// Flush remaining buffer
|
||||
if (buf.trim()) {
|
||||
collectedLines.push(buf);
|
||||
}
|
||||
|
||||
// Same orphan hazard as stdout: a grandchild holding stderr open would
|
||||
// block this drain forever. Race against child exit + a short grace window
|
||||
// (ported from session-runner.ts — the gemini copy lacked it).
|
||||
const stderr = await Promise.race([
|
||||
stderrPromise,
|
||||
(async () => {
|
||||
await procExited;
|
||||
await new Promise((r) => setTimeout(r, 5_000));
|
||||
return '';
|
||||
})(),
|
||||
]);
|
||||
const exitCode = await procExited;
|
||||
clearTimeout(timeoutId);
|
||||
|
||||
const durationMs = Date.now() - startTime;
|
||||
|
||||
// Parse all collected JSONL lines
|
||||
const parsed = parseGeminiJSONL(collectedLines);
|
||||
|
||||
// Log stderr if non-empty (may contain auth errors, etc.)
|
||||
if (stderr.trim()) {
|
||||
process.stderr.write(` [gemini stderr] ${stderr.trim().slice(0, 200)}\n`);
|
||||
}
|
||||
|
||||
// Environment-unusable classification: these are Google-side conditions no
|
||||
// test assertion can act on — the deprecated individual code-assist auth
|
||||
// path ("migrate to the Antigravity suite") and argv drift on older/newer
|
||||
// CLIs. Return the same SKIP shape as binary-not-found so callers report
|
||||
// SKIPPED instead of a false FAIL.
|
||||
const unusableMarkers = [
|
||||
'no longer supported for Gemini Code Assist',
|
||||
'antigravity',
|
||||
'Unknown arguments: skip-trust',
|
||||
];
|
||||
if (exitCode !== 0 && parsed.tokens === 0) {
|
||||
const marker = unusableMarkers.find((m) => stderr.toLowerCase().includes(m.toLowerCase()));
|
||||
if (marker) {
|
||||
return {
|
||||
output: `SKIP: gemini CLI unusable (${marker})`,
|
||||
toolCalls: [],
|
||||
tokens: 0,
|
||||
exitCode: -1,
|
||||
durationMs,
|
||||
sessionId: null,
|
||||
rawLines: collectedLines,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
output: parsed.output,
|
||||
toolCalls: parsed.toolCalls,
|
||||
tokens: parsed.tokens,
|
||||
exitCode: timedOut ? 124 : exitCode,
|
||||
durationMs,
|
||||
sessionId: parsed.sessionId,
|
||||
rawLines: collectedLines,
|
||||
};
|
||||
}
|
||||
@@ -21,6 +21,4 @@ export const OVERLAY_CASE_FILES: Record<string, string> = {
|
||||
'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts': 'opus-4-7-effort-match-trivial',
|
||||
'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts': 'opus-4-7-literal-interpretation',
|
||||
'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts': 'claude-dedicated-tools-vs-bash-sonnet',
|
||||
'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial-sonnet.test.ts': 'opus-4-7-effort-match-trivial-sonnet',
|
||||
'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation-sonnet.test.ts': 'opus-4-7-literal-interpretation-sonnet',
|
||||
};
|
||||
@@ -11,8 +11,8 @@ import { matchGlob } from './touchfiles';
|
||||
|
||||
/** The exact globs package.json's `test:gate` passes to `bun test`. */
|
||||
export const PAID_TEST_GLOBS = [
|
||||
// skill-llm-eval* (not just the base file): skill-llm-eval-spec.test.ts
|
||||
// fell outside the exact glob and could never run in any lane.
|
||||
// skill-llm-eval* (not just the base file): a sibling judge file outside
|
||||
// the exact glob could never run in any lane.
|
||||
'test/skill-llm-eval*.test.ts',
|
||||
'test/skill-e2e-*.test.ts',
|
||||
'test/skill-routing-e2e.test.ts',
|
||||
@@ -22,7 +22,6 @@ export const PAID_TEST_GLOBS = [
|
||||
// entered the paid census. The same bug class as the deleted pre-split
|
||||
// monolith (see test/paid-shards.test.ts's regression pin).
|
||||
'test/codex-e2e*.test.ts',
|
||||
'test/gemini-e2e.test.ts',
|
||||
'test/llm-judge-recommendation.test.ts',
|
||||
'test/carve-section-loading*.test.ts',
|
||||
] as const;
|
||||
|
||||
@@ -15,20 +15,32 @@
|
||||
* the re-entry mechanism.
|
||||
*/
|
||||
export const PERIODIC_CI_EXCLUDE: Record<string, { reason: string; tracking: string }> = {
|
||||
'test/skill-e2e-ship-idempotency.test.ts': {
|
||||
reason:
|
||||
'documented-red: the PTY child sits at the Claude Code welcome screen for the full budget '
|
||||
+ '(readiness/typing race vs CLI 2.1.x); never green since it was born in v1.63',
|
||||
tracking: 'TODOS.md "periodic tier — three documented-red tests need structural repair" (1 of 3 resolved: sidebar trio already deleted)',
|
||||
'test/codex-e2e.test.ts': {
|
||||
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
|
||||
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
|
||||
},
|
||||
'test/skill-e2e-brain-privacy-gate.test.ts': {
|
||||
reason:
|
||||
'documented-red: the artifacts-sync stop-gate preconditions do not survive the hermetic env '
|
||||
+ 'even with per-test HOME/GSTACK_HOME injection; never green anywhere',
|
||||
tracking: 'TODOS.md "periodic tier — three documented-red tests need structural repair"',
|
||||
'test/codex-e2e-sol-scope.test.ts': {
|
||||
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
|
||||
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
|
||||
},
|
||||
'test/skill-e2e-ios.test.ts': {
|
||||
reason: 'requires a live iOS device/simulator toolchain (xcodebuild, devicectl) — manual hardware, not a CI runner capability',
|
||||
tracking: 'TODOS.md "skill-e2e-ios CI story" (device/runner decision)',
|
||||
'test/codex-e2e-shared-libs.test.ts': {
|
||||
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
|
||||
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
|
||||
},
|
||||
'test/codex-e2e-recommendation-substance.test.ts': {
|
||||
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
|
||||
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
|
||||
},
|
||||
'test/skill-e2e-outside-voice.test.ts': {
|
||||
reason: 'needs both the claude and codex CLIs; codex is not in the CI image, so every case self-skips',
|
||||
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
|
||||
},
|
||||
'test/skill-e2e-aside.test.ts': {
|
||||
reason: 'needs macOS with the Aside app open (asideAvailable()); CI runners are Linux, so every case self-skips',
|
||||
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
|
||||
},
|
||||
'test/skill-e2e-ios-device.test.ts': {
|
||||
reason: 'needs a physical iPhone over USB/devicectl — manual hardware, not a CI runner capability',
|
||||
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
|
||||
},
|
||||
};
|
||||
@@ -3,8 +3,8 @@ import * as path from 'node:path';
|
||||
/**
|
||||
* Provider adapter interface — uniform contract for Claude, GPT, Gemini.
|
||||
*
|
||||
* Each adapter wraps an existing runner (session-runner.ts, codex-session-runner.ts,
|
||||
* gemini-session-runner.ts) and normalizes its per-provider result shape into the
|
||||
* Each adapter wraps an existing runner or CLI (session-runner.ts,
|
||||
* codex-session-runner.ts, the gemini CLI) and normalizes its per-provider result shape into the
|
||||
* RunResult below. The benchmark harness only talks to adapters through this
|
||||
* interface, never to the underlying runners directly.
|
||||
*/
|
||||
|
||||
@@ -334,23 +334,6 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
||||
// Conductor → prose decision brief (Conductor signal makes prose the default;
|
||||
// the PreToolUse hook denies the flaky tool). Touches the resolver that owns
|
||||
// the Conductor rule, the preamble signal, the hook, and the detection helper.
|
||||
'conductor-prose': [
|
||||
'lib/claude-public-transcript.ts',
|
||||
'test/auto-decide-recommendation-scope.test.ts',
|
||||
'test/fixtures/auto-decide-recommendation-361c.json',
|
||||
'test/auto-decide-target-identity.test.ts',
|
||||
'test/fixtures/auto-decide-target-361c.json',
|
||||
|
||||
'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json',
|
||||
'test/helpers/plan-count-transcript.ts', 'test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures/plan-count-cross-cwd-ancestry-0bcd.json', 'test/plan-count-session-cwd.test.ts',
|
||||
'scripts/resolvers/learnings.ts',
|
||||
"test/plan-scope-recovery-av.test.ts",
|
||||
"test/fixtures/plan-scope-recovery-av.json", 'test/pty-screen-unicode-ap.test.ts', 'test/eng-scope-entry-ap.test.ts',
|
||||
'test/conductor-prose-observation-ao.test.ts', 'test/fixtures/conductor-prose-ao.json',
|
||||
'test/auto-decide-saved-ai.test.ts', 'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'bin/gstack-session-kind', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-preamble-bash.ts', 'scripts/resolvers/preamble.ts', 'plan-eng-review/**', 'hosts/claude/hooks/question-preference-hook.ts', 'hosts/claude/hooks/spawned-directive.ts', 'lib/is-conductor.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-conductor-prose.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**',
|
||||
"test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-completion-status.ts",
|
||||
'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/helpers/owned-claude-transcript.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/testing.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts'
|
||||
],
|
||||
|
||||
// Native question capture and interactive workflow probes.
|
||||
'auq-format-gate': ['test/session-runner-stream-lifecycle.test.ts', 'plan-ceo-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completeness-section.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/session-runner.ts', 'test/helpers/llm-judge.ts', 'test/skill-e2e-ask-user-question-format-compliance.test.ts',
|
||||
@@ -393,9 +376,6 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
||||
"test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts",
|
||||
'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/plan-design-with-ui-fixture.test.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/helpers/plan-review-cases.ts', 'test/helpers/plan-review-board-feedback.ts', 'test/plan-review-board-feedback.test.ts', 'test/fixtures/design-board-questions.json', 'test/fixtures/design-outside-voices-question.json', 'design/src/daemon-state.ts', 'design/src/daemon.ts', 'design/test/daemon-tests-fixtures.ts', 'design/src/daemon-client.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'bin/gstack-paths', 'bin/gstack-slug', 'scripts/resolvers/design.ts'
|
||||
],
|
||||
'ship-idempotency-pty': ['ship/**', 'bin/gstack-next-version', 'bin/gstack-version-bump', 'scripts/resolvers/sections.ts', 'lib/worktree.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-ship-idempotency.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt',
|
||||
'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/testing.ts'
|
||||
],
|
||||
'tpa-present': ['test/session-runner-stream-lifecycle.test.ts', 'scripts/resolvers/third-party-actions.ts', 'ship/SKILL.md.tmpl', 'ship/sections/apple-release.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-third-party-actions.test.ts', 'test/helpers/third-party-actions.ts',
|
||||
'test/third-party-actions-recording.test.ts', 'lib/eval-model.ts'
|
||||
],
|
||||
@@ -784,9 +764,6 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
||||
'test/fixtures/plan-count-quoted-frame-ak.json',
|
||||
'docs/askuserquestion-split.md', 'test/resolver-ask-user-format.test.ts', 'bin/gstack-slug', 'test/helpers/hermetic-env.test.ts', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/ceo-finding-fixture.test.ts', 'test/helpers/owned-claude-transcript.ts', 'test/eval-budgets-policy.test.ts', 'test/fixtures/webfetch-permission.json', 'test/plan-skill-webfetch-permission.test.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'test/helpers/plan-review-decisions.ts', 'test/plan-review-decisions.test.ts', 'test/helpers/plan-review-cases.ts', 'test/plan-review-cases.test.ts', 'test/helpers/llm-judge.ts', 'lib/eval-model.ts', 'test/skill-e2e-plan-decision-classification.test.ts', 'test/fixtures/plan-decision-classification.ts', 'test/plan-review-calibration.test.ts', 'test/helpers/ceo-split-question-policy.ts', 'test/fixtures/ceo-split-actor-6aef.json', 'test/helpers/ceo-mode-option.ts', 'test/ceo-mode-option.test.ts', 'test/ceo-split-collection.test.ts', 'test/fixtures/ceo-split-collection-0bcd.json', 'test/ceo-split-question-policy.test.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/tasks-section.ts'
|
||||
],
|
||||
'brain-privacy-gate': ['bin/gstack-skill-start', 'bin/gstack-skill-end', 'scripts/resolvers/preamble/generate-brain-sync-block.ts', 'scripts/resolvers/preamble.ts', 'bin/gstack-brain-sync', 'bin/gstack-artifacts-init', 'bin/gstack-config', 'test/helpers/agent-sdk-runner.ts', 'test/skill-e2e-brain-privacy-gate.test.ts',
|
||||
'test/agent-sdk-runner.test.ts'
|
||||
],
|
||||
|
||||
// /setup-gbrain Path 4 (Remote MCP) — happy + bad-token end-to-end via
|
||||
// Agent SDK. Gate-tier (deterministic stub server, fixed inputs); fires
|
||||
@@ -864,11 +841,6 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
||||
'plan-tune-inspect': ['test/session-runner-stream-lifecycle.test.ts', 'plan-tune/**', 'scripts/question-registry.ts', 'scripts/psychographic-signals.ts', 'scripts/one-way-doors.ts', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'bin/gstack-developer-profile', 'test/skill-e2e-plan-tune.test.ts'],
|
||||
|
||||
// /plan-tune cathedral (T16 — 5 E2E scenarios, all gate per D12)
|
||||
'plan-tune-hook-capture': ['hosts/claude/hooks/**', 'bin/gstack-question-log', 'bin/gstack-developer-profile', 'plan-tune/**', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'],
|
||||
'plan-tune-enforcement': ['hosts/claude/hooks/**', 'bin/gstack-question-preference', 'scripts/question-registry.ts', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'],
|
||||
'plan-tune-annotation': ['hosts/claude/hooks/**', 'scripts/declared-annotation.ts', 'scripts/psychographic-signals.ts', 'scripts/question-registry.ts', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'],
|
||||
'plan-tune-codex-import': ['bin/gstack-codex-session-import', 'bin/gstack-question-log', 'docs/spikes/codex-session-format.md', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'],
|
||||
'plan-tune-dream-cycle': ['bin/gstack-distill-free-text', 'bin/gstack-distill-apply', 'hosts/claude/hooks/**', 'plan-tune/**', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'],
|
||||
|
||||
// Codex offering verification
|
||||
'codex-offered-office-hours': ['test/session-runner-stream-lifecycle.test.ts', 'test/paid-retry-supervision.test.ts', 'office-hours/**', 'scripts/gen-skill-docs.ts', 'test/skill-e2e-plan.test.ts',
|
||||
@@ -1013,7 +985,6 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
||||
],
|
||||
|
||||
// Gemini E2E — smoke test only (Gemini gets lost in worktrees on complex tasks)
|
||||
'gemini-smoke': ['scripts/gen-skill-docs.ts', 'test/helpers/gemini-session-runner.ts', 'lib/worktree.ts', 'test/gemini-e2e.test.ts'],
|
||||
|
||||
|
||||
// Coverage audit (shared fixture) + triage + gates
|
||||
@@ -1264,15 +1235,16 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
||||
|
||||
"test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts",
|
||||
],
|
||||
'journey-negatives': ['test/session-runner-stream-lifecycle.test.ts',
|
||||
'test/skill-fixture.test.ts',
|
||||
"test/plan-scope-recovery-av.test.ts",
|
||||
"test/fixtures/plan-scope-recovery-av.json",
|
||||
"test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts',
|
||||
"test/design-scope-entry-aq.test.ts",
|
||||
|
||||
"test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts",
|
||||
],
|
||||
|
||||
// Opus 4.7 behavior evals — keys match testName: values in the test file.
|
||||
// Routing sub-tests use template literal `routing-${c.name}` testNames,
|
||||
// which the touchfile completeness scanner skips; they inherit selection
|
||||
// from the file-level touchfile entry via GLOBAL_TOUCHFILES.
|
||||
'fanout-arm-overlay-on':
|
||||
['test/session-runner-stream-lifecycle.test.ts', 'model-overlays/claude.md', 'model-overlays/opus-4-7.md', 'scripts/models.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-opus-47.test.ts'],
|
||||
'fanout-arm-overlay-off':
|
||||
['test/session-runner-stream-lifecycle.test.ts', 'model-overlays/claude.md', 'model-overlays/opus-4-7.md', 'scripts/models.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-opus-47.test.ts'],
|
||||
|
||||
// Overlay efficacy harness (SDK) — measures whether overlay nudges change
|
||||
// behavior under @anthropic-ai/claude-agent-sdk (closer to real Claude Code
|
||||
@@ -1280,22 +1252,10 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
||||
// completeness scanner doesn't require them; these entries exist for
|
||||
// diff-based selection accuracy.
|
||||
|
||||
// /ios-qa — agent flow E2E. Daemon + stub StateServer + codegen
|
||||
// exercised end-to-end. The no-device path is gate-tier; the with-device
|
||||
// path requires GSTACK_HAS_IOS_DEVICE=1 and is periodic-tier.
|
||||
'ios-qa-e2e': ['ios-qa/**', 'ios-fix/**', 'ios-design-review/**', 'ios-clean/**', 'ios-sync/**', 'test/skill-e2e-ios.test.ts'],
|
||||
// Swift-build invariant test — requires the Swift toolchain. Compiles the
|
||||
// fixture SPM package + runs the XCTest suite that validates the real
|
||||
// Swift StateServer implementation (loopback bind, boot token rotation,
|
||||
// session lock). Periodic-tier — Swift build is heavier than TS unit tests.
|
||||
'ios-qa-swift-build': ['ios-qa/templates/**', 'test/fixtures/ios-qa/FixtureApp/**', 'test/skill-e2e-ios-swift-build.test.ts'],
|
||||
// Real-device path — only runs with GSTACK_HAS_IOS_DEVICE=1 + a paired
|
||||
// iPhone. Validates the CoreDevice agent + iOS SDK toolchain. Periodic-tier.
|
||||
'ios-qa-device': ['ios-qa/templates/**', 'test/fixtures/ios-qa/FixtureApp/**', 'test/skill-e2e-ios-device.test.ts'],
|
||||
|
||||
// /spec end-to-end via PTY — exercises the full Phase 1→5 pipeline
|
||||
// including --execute spawn. Periodic-tier — paid + non-deterministic.
|
||||
'spec-execute': ['spec/**', 'scripts/resolvers/redact-doc.ts', 'scripts/resolvers/outside-voice.ts', 'scripts/resolvers/constants.ts', 'lib/outside-review-result.ts', 'bin/gstack-redact', 'lib/redact-engine.ts', 'test/skill-e2e-spec-execute.test.ts'],
|
||||
|
||||
// /office-hours brain-writeback path under fake gbrain CLI (v1.50.0.0
|
||||
// T7). Drives /office-hours with a regenerated SKILL.md that has the
|
||||
@@ -1370,16 +1330,10 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
||||
'test/fixtures/eng-omitted-select-361c.json',
|
||||
'test/skill-e2e-plan-decision-classification.test.ts', 'test/fixtures/plan-decision-classification.ts', 'test/plan-review-calibration.test.ts', 'test/helpers/plan-review-decisions.ts', 'test/plan-review-decisions.test.ts', 'test/helpers/plan-review-cases.ts', 'test/plan-review-cases.test.ts', 'test/helpers/llm-judge.ts', 'lib/eval-model.ts', 'test/helpers/e2e-helpers.ts', 'test/helpers/eval-store.ts', 'test/helpers/eval-budgets.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'docs/askuserquestion-split.md', 'plan-ceo-review/SKILL.md.tmpl', 'plan-ceo-review/sections/review-sections.md.tmpl', 'scripts/resolvers/tasks-section.ts'],
|
||||
'health-reporting': ['health/**', 'test/skill-e2e-health.test.ts', 'test/helpers/health-eval-fixture.ts'],
|
||||
'codex-plan-ceo-format-mode': ['test/paid-retry-supervision.test.ts', 'plan-ceo-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts'],
|
||||
'codex-plan-ceo-format-approach': ['test/paid-retry-supervision.test.ts', 'plan-ceo-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts'],
|
||||
'codex-plan-eng-format-coverage': ['test/paid-retry-supervision.test.ts', 'scripts/resolvers/learnings.ts', 'test/review-entry-and-design-clarity-au.test.ts', 'test/fixtures/plan-scope-recovery-av.json', 'test/plan-scope-recovery-av.test.ts', 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts', 'test/plan-review-cases.test.ts'],
|
||||
'codex-plan-eng-format-kind': ['test/paid-retry-supervision.test.ts', 'scripts/resolvers/learnings.ts', 'test/review-entry-and-design-clarity-au.test.ts', 'test/fixtures/plan-scope-recovery-av.json', 'test/plan-scope-recovery-av.test.ts', 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts', 'test/plan-review-cases.test.ts'],
|
||||
'overlay-harness-claude-dedicated-tools-vs-bash': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
|
||||
'overlay-harness-opus-4-7-effort-match-trivial': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
|
||||
'overlay-harness-opus-4-7-literal-interpretation': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
|
||||
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
|
||||
'overlay-harness-opus-4-7-effort-match-trivial-sonnet': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial-sonnet.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
|
||||
'overlay-harness-opus-4-7-literal-interpretation-sonnet': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation-sonnet.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'],
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -1503,7 +1457,6 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
// v1.21+ auto-mode regression tests
|
||||
'office-hours-auto-mode': 'gate',
|
||||
'auto-decide-preserved': 'periodic',
|
||||
'conductor-prose': 'periodic',
|
||||
|
||||
// Real-PTY E2E batch — tier classification:
|
||||
// gate: cheap, deterministic, run on every PR
|
||||
@@ -1511,7 +1464,6 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
'auq-format-gate': 'gate', // ~$0.50/run, native SDK question capture, single skill probe
|
||||
'plan-ceo-mode-routing': 'periodic', // ~$3/run, deep navigation through 8-12 prior AskUserQuestions
|
||||
'plan-design-with-ui-scope': 'gate', // ~$0.80/run
|
||||
'ship-idempotency-pty': 'periodic', // ~$3/run, real /ship in plan mode
|
||||
'tpa-present': 'gate', // consent/credential safety guardrail; deterministic shims + grep asserts
|
||||
'tpa-absent-linux': 'gate', // consent/credential safety guardrail; deterministic shims + grep asserts
|
||||
'tpa-broken': 'gate', // consent/credential safety guardrail; deterministic shims + grep asserts
|
||||
@@ -1540,7 +1492,6 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
|
||||
// Privacy gate for gstack-brain-sync — periodic (non-deterministic LLM call,
|
||||
// costs ~$0.30-$0.50 per run, not needed on every commit)
|
||||
'brain-privacy-gate': 'periodic',
|
||||
|
||||
// /setup-gbrain Path 4 (Remote MCP) — periodic-tier. The stub HTTP
|
||||
// server is deterministic but the model's interpretation of "follow
|
||||
@@ -1578,11 +1529,6 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
'plan-tune-inspect': 'gate',
|
||||
|
||||
// /plan-tune cathedral (T16 per D12 — all gate)
|
||||
'plan-tune-hook-capture': 'gate',
|
||||
'plan-tune-enforcement': 'gate',
|
||||
'plan-tune-annotation': 'gate',
|
||||
'plan-tune-codex-import': 'gate',
|
||||
'plan-tune-dream-cycle': 'gate',
|
||||
|
||||
// Codex offering verification
|
||||
'codex-offered-office-hours': 'gate',
|
||||
@@ -1646,7 +1592,6 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
'outside-voice-claude-code-to-codex': 'periodic',
|
||||
'outside-plan-disabled-no-fallback': 'periodic',
|
||||
'codex-sol-scope-termination': 'periodic',
|
||||
'gemini-smoke': 'periodic',
|
||||
|
||||
// Design — gate for cheap functional, periodic for Opus/quality
|
||||
'design-consultation-core': 'periodic',
|
||||
@@ -1706,10 +1651,9 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
'journey-retro': 'periodic',
|
||||
'journey-design-system': 'periodic',
|
||||
'journey-visual-qa': 'periodic',
|
||||
'journey-negatives': 'periodic',
|
||||
|
||||
// Opus 4.7 overlay evals — periodic (non-deterministic LLM behavior + Opus cost)
|
||||
'fanout-arm-overlay-on': 'periodic',
|
||||
'fanout-arm-overlay-off': 'periodic',
|
||||
|
||||
// Overlay efficacy harness (SDK, paid) — periodic only
|
||||
|
||||
@@ -1720,13 +1664,10 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
// on every Linux PR. Periodic keeps it in the weekly census on capable
|
||||
// hosts; re-promote if a macOS runner lands (flagged decision in the
|
||||
// test-infra overhaul plan).
|
||||
'ios-qa-e2e': 'periodic',
|
||||
// Swift toolchain only, no device required, but heavier than TS unit tests.
|
||||
'ios-qa-swift-build': 'periodic',
|
||||
// Requires a real connected + paired iPhone. Manual-trigger only.
|
||||
'ios-qa-device': 'periodic',
|
||||
// /spec end-to-end PTY pipeline (paid, non-deterministic — periodic-tier).
|
||||
'spec-execute': 'periodic',
|
||||
|
||||
// WS2 arm benchmark — periodic: full build-shaped agentic workflows, paid,
|
||||
// non-deterministic by construction (research instrument, not a gate).
|
||||
@@ -1737,16 +1678,10 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
'plan-decision-classification': 'periodic',
|
||||
'plan-devex-peer-comparison-classification': 'periodic',
|
||||
'health-reporting': 'periodic',
|
||||
'codex-plan-ceo-format-mode': 'periodic',
|
||||
'codex-plan-ceo-format-approach': 'periodic',
|
||||
'codex-plan-eng-format-coverage': 'periodic',
|
||||
'codex-plan-eng-format-kind': 'periodic',
|
||||
'overlay-harness-claude-dedicated-tools-vs-bash': 'periodic',
|
||||
'overlay-harness-opus-4-7-effort-match-trivial': 'periodic',
|
||||
'overlay-harness-opus-4-7-literal-interpretation': 'periodic',
|
||||
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'periodic',
|
||||
'overlay-harness-opus-4-7-effort-match-trivial-sonnet': 'periodic',
|
||||
'overlay-harness-opus-4-7-literal-interpretation-sonnet': 'periodic',
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -1754,16 +1689,12 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
*/
|
||||
export const LLM_JUDGE_TOUCHFILES: Record<string, string[]> = {
|
||||
'setup-browser-cookies/SKILL.md workflow': ['setup-browser-cookies/SKILL.md.tmpl', 'setup-browser-cookies/SKILL.md', 'BROWSER.md', 'test/helpers/cookie-workflow-judge-input.ts', 'test/cookie-workflow-judge-input.test.ts', 'test/helpers/cookie-workflow-manual-review.ts', 'test/cookie-workflow-manual-review.test.ts', 'test/helpers/manual-judge-review-fixture.ts', '.github/cookie-workflow-manual-review.json', 'test/helpers/workflow-judge-input.ts', 'test/skill-llm-eval.test.ts'],
|
||||
'command reference table': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/commands.ts', 'gstack/llms.txt', 'test/skill-llm-eval.test.ts'],
|
||||
'snapshot flags reference': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/snapshot.ts', 'test/skill-llm-eval.test.ts'],
|
||||
'browse/SKILL.md reference': ['browse/sections/**', 'browse/SKILL.md', 'browse/SKILL.md.tmpl', 'browse/src/**', 'test/skill-llm-eval.test.ts'],
|
||||
'browse/SKILL.md reference': ['browse/sections/**', 'browse/SKILL.md', 'browse/SKILL.md.tmpl', 'browse/src/**', 'SKILL.md', 'SKILL.md.tmpl', 'gstack/llms.txt', 'test/fixtures/eval-baselines.json', 'test/skill-llm-eval.test.ts'],
|
||||
'setup block': ['browse/SKILL.md', 'browse/SKILL.md.tmpl', 'scripts/resolvers/aside.ts', 'scripts/resolvers/browse.ts', 'test/skill-llm-eval.test.ts'],
|
||||
'regression vs baseline': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/commands.ts', 'test/fixtures/eval-baselines.json', 'test/skill-llm-eval.test.ts'],
|
||||
'qa/SKILL.md workflow': ['qa/sections/**', 'qa/SKILL.md', 'qa/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'],
|
||||
'qa/SKILL.md health rubric': ['qa/sections/**', 'qa/SKILL.md', 'qa/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'],
|
||||
'qa/SKILL.md anti-refusal': ['qa/sections/**', 'qa/SKILL.md', 'qa/SKILL.md.tmpl', 'qa-only/SKILL.md', 'qa-only/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'],
|
||||
'cross-skill greptile consistency': ['review/SKILL.md', 'review/SKILL.md.tmpl', 'ship/SKILL.md', 'ship/SKILL.md.tmpl', 'review/greptile-triage.md', 'retro/SKILL.md', 'retro/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'],
|
||||
'baseline score pinning': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'test/fixtures/eval-baselines.json', 'test/skill-llm-eval.test.ts'],
|
||||
|
||||
// Ship & Release
|
||||
'ship/SKILL.md workflow': ['ship/SKILL.md', 'ship/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts', 'test/helpers/workflow-judge-input.ts', 'test/helpers/workflow-judge-cache.ts', 'test/workflow-judge-cache.test.ts', 'scripts/eval-input-cache.ts', 'test/eval-input-cache.test.ts', 'test/workflow-judge-input.test.ts', 'test/helpers/workflow-excerpt.ts',
|
||||
|
||||
@@ -25,7 +25,6 @@ const RUNNERS = [
|
||||
'test/helpers/session-runner.ts',
|
||||
'test/helpers/claude-pty-runner.ts',
|
||||
'test/helpers/codex-session-runner.ts',
|
||||
'test/helpers/gemini-session-runner.ts',
|
||||
'test/helpers/agent-sdk-runner.ts',
|
||||
];
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// Swift-build invariant tests. Runs against the fixture iOS app at
|
||||
// test/fixtures/ios-qa/FixtureApp/. Requires the Swift toolchain
|
||||
// (Xcode CLI tools or stand-alone Swift). Skipped if swift is not on PATH.
|
||||
// (Xcode CLI tools or stand-alone Swift). The swift build invariants run
|
||||
// only with GSTACK_TEST_SWIFT=1; the parity and harness pins always run.
|
||||
//
|
||||
// Two invariants:
|
||||
//
|
||||
@@ -298,13 +299,9 @@ describe('iOS tap harness regressions', () => {
|
||||
});
|
||||
});
|
||||
|
||||
function hasSwift(): boolean {
|
||||
const r = spawnSync('swift', ['--version'], { stdio: 'pipe', timeout: 30_000 });
|
||||
return r.status === 0;
|
||||
}
|
||||
|
||||
const swiftAvailable = hasSwift();
|
||||
const describeIfSwift = swiftAvailable ? describe : describe.skip;
|
||||
// Explicit opt-in, not tool presence: free shards run on hosts where Swift
|
||||
// may be installed, and a full swift build does not belong in every PR run.
|
||||
const describeIfSwift = process.env.GSTACK_TEST_SWIFT === '1' ? describe : describe.skip;
|
||||
|
||||
describeIfSwift('swift build invariants', () => {
|
||||
// DebugBridgeUI + DebugBridgeTouch are iOS-only (they link UIKit). Plain
|
||||
@@ -1,12 +1,9 @@
|
||||
// High-level E2E for /ios-qa skill flow.
|
||||
//
|
||||
// Two scenarios:
|
||||
// 1. NO_DEVICE (gate-tier compatible): runs the gen-accessors codegen
|
||||
// against a SwiftUI fixture, verifies output is correct, no daemon
|
||||
// hardware required. Catches regression in source-read + codegen +
|
||||
// cache + render paths without an iPhone.
|
||||
// 2. WITH_DEVICE (periodic-tier, requires GSTACK_HAS_IOS_DEVICE=1): full
|
||||
// daemon + tailnet + USB tunnel loop. Skipped in CI.
|
||||
// Runs the gen-accessors codegen against a SwiftUI fixture and simulates the
|
||||
// agent flow against the daemon with a fake device tunnel — no hardware.
|
||||
// Catches regression in source-read + codegen + cache + render paths without
|
||||
// an iPhone. The real-device loop lives in test/skill-e2e-ios-device.test.ts.
|
||||
//
|
||||
// Note: The detailed daemon HTTP unit/integration tests live next to the
|
||||
// daemon source (ios-qa/daemon/test/*). This file tests the agent-flow
|
||||
@@ -22,7 +19,6 @@ import type { DeviceTunnel } from '../ios-qa/daemon/src/proxy';
|
||||
import { grantIdentity } from '../ios-qa/daemon/src/allowlist';
|
||||
import { generate } from '../ios-qa/scripts/gen-accessors';
|
||||
|
||||
const HAS_DEVICE = process.env.GSTACK_HAS_IOS_DEVICE === '1';
|
||||
|
||||
const DEVICE_TOKEN = 'rotated-mock-bearer-token';
|
||||
|
||||
@@ -494,14 +490,3 @@ describe('ios-qa E2E (agent-flow simulation)', () => {
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
// ───────── WITH_DEVICE — manual smoke tests (skipped in CI) ─────────
|
||||
|
||||
(HAS_DEVICE ? describe : describe.skip)('ios-qa E2E (with device)', () => {
|
||||
test('WITH_DEVICE: full agent loop against a real iPhone', () => {
|
||||
const workDir = makeWorkDir();
|
||||
// Stub — real implementation requires `devicectl` + an attached iPhone.
|
||||
// Documented in ios-qa/SKILL.md.tmpl under "Manual smoke test".
|
||||
expect(HAS_DEVICE).toBe(true);
|
||||
});
|
||||
});
|
||||
File renamed without changes.
@@ -98,7 +98,7 @@ test('the native reader cannot promote foreign cwd, child, user or tool-result t
|
||||
});
|
||||
|
||||
test('new native annotation dependencies retain all existing observation caller owners',()=>{
|
||||
const expected=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','office-hours-auto-mode','auto-decide-preserved','conductor-prose'];
|
||||
const expected=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','office-hours-auto-mode','auto-decide-preserved'];
|
||||
for(const file of ['test/helpers/native-auto-decide.ts','test/native-auto-decide.test.ts','test/native-auto-decide-pty.test.ts','test/fixtures/native-auto-decide-ag.json']){
|
||||
const owners=Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(file)).map(([name])=>name);expect(owners).toEqual(expected);
|
||||
expect(selectTests([file],E2E_TOUCHFILES).selected).toContain('auto-decide-preserved');
|
||||
|
||||
@@ -27,8 +27,8 @@ test('every fixture owns one paid wrapper; public models/trials/concurrency/turn
|
||||
}
|
||||
expect(isPaidTestFile('test/overlay-lifecycle.test.ts')).toBe(false);
|
||||
expect(OVERLAY_FIXTURES.every(f => f.trials === 10 && f.concurrency === 3)).toBe(true);
|
||||
expect(OVERLAY_FIXTURES.map(f => f.maxTurns ?? 5)).toEqual([15,8,15,15,8,15]);
|
||||
expect(OVERLAY_FIXTURES.map(f => f.model)).toEqual([...Array(3).fill('claude-opus-4-7'), ...Array(3).fill('claude-sonnet-4-6')]);
|
||||
expect(OVERLAY_FIXTURES.map(f => f.maxTurns ?? 5)).toEqual([15,8,15,15]);
|
||||
expect(OVERLAY_FIXTURES.map(f => f.model)).toEqual([...Array(3).fill('claude-opus-4-7'), 'claude-sonnet-4-6']);
|
||||
expect([OVERLAY_CASE_WORK_MS, OVERLAY_RECORD_GRACE_MS, OVERLAY_CASE_OUTER_MS, OVERLAY_MIN_FILE_WALL_MS]).toEqual([1_800_000, 5_000, 1_810_000, 1_830_000]);
|
||||
});
|
||||
|
||||
|
||||
@@ -226,11 +226,11 @@ describe('literal fixture task correctness', () => {
|
||||
expect(() => assertReadOnlyWorkspace(before, snapshotWorkspace(dir))).toThrow('b');
|
||||
}));
|
||||
test('registry retires absent-nudge fanout cases and preserves remaining model/trial budgets', () => {
|
||||
expect(OVERLAY_FIXTURES).toHaveLength(6);
|
||||
expect(OVERLAY_FIXTURES).toHaveLength(4);
|
||||
expect(OVERLAY_FIXTURES.some((f) => f.id.includes('fanout') || f.comparison?.unsupportedHypothesis)).toBe(false);
|
||||
expect(OVERLAY_FIXTURES.every((f) => f.trials === 10 && f.comparison)).toBe(true);
|
||||
expect(OVERLAY_FIXTURES.filter((f) => f.model === 'claude-opus-4-7')).toHaveLength(3);
|
||||
expect(OVERLAY_FIXTURES.filter((f) => f.model === 'claude-sonnet-4-6')).toHaveLength(3);
|
||||
expect(OVERLAY_FIXTURES.filter((f) => f.model === 'claude-sonnet-4-6')).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -52,7 +52,7 @@ describe('overlay file policy', () => {
|
||||
});
|
||||
|
||||
test('only the exact wrapper family gets one attempt and the extra process grace', () => {
|
||||
expect(overlayFiles).toHaveLength(6);
|
||||
expect(overlayFiles).toHaveLength(4);
|
||||
expect(OVERLAY_MAX_ACTIVE_SHARDS).toBe(1);
|
||||
expect(OVERLAY_MIN_FILE_WALL_MS).toBe(1_830_000);
|
||||
for (const file of overlayFiles) {
|
||||
@@ -123,16 +123,16 @@ describe('overlay file policy', () => {
|
||||
});
|
||||
|
||||
describe('overlay manifest affinity and CI capacity', () => {
|
||||
test('actual lifecycle wrapper guards include all six in periodic and exclude all six from gate', () => {
|
||||
test('actual lifecycle wrapper guards include all four in periodic and exclude all four from gate', () => {
|
||||
for (const tier of ['periodic', 'gate'] as const) {
|
||||
const manifest = buildRunManifest({ tier, sliceCount: 6, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
const entries = manifest.entries.filter(entry => isOverlayTestFile(entry.file));
|
||||
expect(entries).toHaveLength(6);
|
||||
expect(entries).toHaveLength(4);
|
||||
expect(entries.every(entry => entry.status === (tier === 'periodic' ? 'planned' : 'excluded'))).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
test('95 files retain every case, reserve slice six, and fit 330 minutes with actual family walls', () => {
|
||||
test('93 files retain every case, reserve slice six, and fit 330 minutes with actual family walls', () => {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'overlay-affinity-'));
|
||||
const normalFiles = Array.from({ length: 89 }, (_, i) => `test/skill-e2e-normal-${i.toString().padStart(2, '0')}.test.ts`);
|
||||
const discovered = [...normalFiles, ...overlayFiles];
|
||||
@@ -144,11 +144,11 @@ describe('overlay manifest affinity and CI capacity', () => {
|
||||
}
|
||||
const opts = { tier: 'periodic' as const, sliceCount: 6, evalsAll: true, discovered, rootDir: dir, env: { EVALS_ALL: '1' } };
|
||||
const manifest = buildRunManifest(opts);
|
||||
expect(manifest.entries).toHaveLength(95);
|
||||
expect(new Set(manifest.entries.map(e => e.file)).size).toBe(95);
|
||||
expect(manifest.entries).toHaveLength(93);
|
||||
expect(new Set(manifest.entries.map(e => e.file)).size).toBe(93);
|
||||
expect(manifest.entries.every(e => e.status === 'planned')).toBe(true);
|
||||
const counts = [1, 2, 3, 4, 5, 6].map(slice => manifest.entries.filter(e => e.slice === slice).length);
|
||||
expect(counts).toEqual([18, 18, 18, 18, 17, 6]);
|
||||
expect(counts).toEqual([18, 18, 18, 18, 17, 4]);
|
||||
expect(manifest.entries.filter(e => e.slice === 6).map(e => e.file).sort()).toEqual([...overlayFiles].sort());
|
||||
expect(buildRunManifest({ ...opts, discovered: [...discovered].reverse() })).toEqual(manifest);
|
||||
expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest);
|
||||
@@ -172,7 +172,7 @@ describe('overlay manifest affinity and CI capacity', () => {
|
||||
const overlayMinutes = Math.ceil(overlayFiles.length / OVERLAY_MAX_ACTIVE_SHARDS)
|
||||
* Math.max(...overlayFiles.map(file => resolvePaidShardTimeoutMs([file]))) / 60_000;
|
||||
expect(normalMinutes).toBe(270);
|
||||
expect(overlayMinutes).toBe(183);
|
||||
expect(overlayMinutes).toBe(122);
|
||||
expect(job['timeout-minutes']).toBeGreaterThanOrEqual(Math.max(normalMinutes, overlayMinutes) + 20);
|
||||
expect(job['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(AUTOPLAN_CHAIN_BUDGET.ciJobMs);
|
||||
|
||||
|
||||
@@ -18,7 +18,7 @@ import { resolveModuleSelection } from './helpers/e2e-helpers';
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const CEO_FILES = [
|
||||
'test/skill-e2e-plan.test.ts', 'test/skill-e2e-ask-user-question-format-compliance.test.ts',
|
||||
'test/skill-e2e-opus-47.test.ts', 'test/skill-llm-eval.test.ts',
|
||||
'test/skill-e2e-retro.test.ts', 'test/skill-llm-eval.test.ts',
|
||||
];
|
||||
const ceoManifest = () => buildRunManifest({ tier: 'gate', profile: 'pr', sliceCount: 1,
|
||||
evalsAll: false, env: {}, changedFiles: ['plan-ceo-review/SKILL.md.tmpl'], discovered: CEO_FILES });
|
||||
@@ -70,7 +70,7 @@ describe('PR profile paid-runner integration', () => {
|
||||
expect(manifest.selection?.e2e).toContain('plan-ceo-review-benefits');
|
||||
expect(manifest.selection?.e2e).not.toContain('plan-ceo-review-plan-mode');
|
||||
expect(manifest.selection?.judges).toContain('plan-ceo-review/SKILL.md modes');
|
||||
expect(manifest.entries.find(entry => entry.file.includes('opus-47'))?.status).toBe('skipped-by-diff');
|
||||
expect(manifest.entries.find(entry => entry.file.includes('skill-e2e-retro'))?.status).toBe('skipped-by-diff');
|
||||
expect(manifest.entries.find(entry => entry.file.includes('ask-user-question'))?.status).toBe('planned');
|
||||
expect(manifest.prCoverage?.deferred.some(item => item.id === 'plan-ceo-review-plan-mode')).toBe(true);
|
||||
expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest);
|
||||
@@ -145,8 +145,8 @@ describe('PR profile paid-runner integration', () => {
|
||||
broad.selection!.e2e!.push('qa-only-no-fix'); broad.prCoverage!.e2e.push('qa-only-no-fix');
|
||||
expect(() => parseRunManifest(JSON.stringify(broad))).toThrow('broad-only');
|
||||
const injected = structuredClone(manifest);
|
||||
injected.entries.find(entry => entry.file.includes('opus-47'))!.status = 'planned';
|
||||
injected.entries.find(entry => entry.file.includes('opus-47'))!.slice = 1;
|
||||
injected.entries.find(entry => entry.file.includes('skill-e2e-retro'))!.status = 'planned';
|
||||
injected.entries.find(entry => entry.file.includes('skill-e2e-retro'))!.slice = 1;
|
||||
expect(() => parseRunManifest(JSON.stringify(injected))).toThrow('outside its PR case selection');
|
||||
for (const action of ['remove', 'skip', 'duplicate'] as const) {
|
||||
const missing = structuredClone(manifest);
|
||||
@@ -184,7 +184,7 @@ describe('PR profile paid-runner integration', () => {
|
||||
const judge = new RegExp(prProfileTestNamePattern('test/skill-llm-eval.test.ts', selection));
|
||||
expect(judge.test('LLM-as-judge plan-ceo-review/SKILL.md modes')).toBe(true);
|
||||
expect(judge.test('LLM-as-judge plan-ceo-review/SKILLxmd modes')).toBe(false);
|
||||
expect(prProfileFileSelected('test/skill-e2e-opus-47.test.ts', selection)).toBe(false);
|
||||
expect(prProfileFileSelected('test/skill-e2e-retro.test.ts', selection)).toBe(false);
|
||||
});
|
||||
|
||||
test('real Bun child executes only the persisted case through the actual registered helper', async () => {
|
||||
|
||||
@@ -13,9 +13,8 @@ import {
|
||||
const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8');
|
||||
const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file));
|
||||
const expectedWalls = {
|
||||
'test/skill-llm-eval.test.ts': 6_920_000,
|
||||
'test/skill-llm-eval.test.ts': 5_960_000,
|
||||
'test/skill-e2e-auq-consistency.test.ts': 2_040_000,
|
||||
'test/codex-e2e-plan-format.test.ts': 5_000_000,
|
||||
'test/skill-e2e-auq-matrix.test.ts': 3_720_000,
|
||||
'test/skill-e2e-plan-format.test.ts': 2_600_000,
|
||||
'test/skill-e2e-auto-decide-preserved.test.ts': 1_920_000,
|
||||
@@ -30,9 +29,9 @@ const expectedWalls = {
|
||||
'test/skill-e2e-plan.test.ts': 7_320_000,
|
||||
};
|
||||
|
||||
test('registration covers exactly the fifteen demonstrated full-file retry gaps', () => {
|
||||
test('registration covers exactly the fourteen demonstrated full-file retry gaps', () => {
|
||||
expect(Object.fromEntries(newBudgets.map(row => [row.file, row.shardMs]))).toEqual(expectedWalls);
|
||||
expect(new Set(FILE_RETRY_BUDGETS.map(row => row.file)).size).toBe(21);
|
||||
expect(new Set(FILE_RETRY_BUDGETS.map(row => row.file)).size).toBe(20);
|
||||
expect(STRICT_RETRY_CASE_BUDGETS.map(row => row.file)).toEqual([
|
||||
...FINDING_RETRY_BUDGETS.map(row => row.file), AUQ_CONSISTENCY_RETRY_BUDGET.file,
|
||||
]);
|
||||
@@ -51,8 +50,6 @@ test('source allowances retain all captures, cases, and finalization grace', ()
|
||||
expect(auq).toContain('for (const [i, capture] of captures.entries())');
|
||||
expect(auq).toContain('N_RUNS * CAPTURE_MS + 60_000');
|
||||
expect(AUQ_CONSISTENCY_RETRY_BUDGET.testMs).toBe(960_000);
|
||||
expect(timeoutExpressions('test/codex-e2e-plan-format.test.ts')).toEqual(Array(4).fill('CAPTURE_LONG_MS'));
|
||||
expect(read('test/codex-e2e-plan-format.test.ts')).toContain('test(name, fn, timeout + CODEX_EVAL_FINALIZE_MS)');
|
||||
expect(read('test/helpers/codex-eval.ts')).toContain('CODEX_EVAL_FINALIZE_MS = 2 * CODEX_DRAIN_GRACE_MS');
|
||||
expect(read('test/helpers/codex-session-runner.ts')).toMatch(/CODEX_DRAIN_GRACE_MS\s*=\s*5_000/);
|
||||
expect(read('test/helpers/office-hours-attempt.ts')).toContain('OFFICE_HOURS_BUN_GRACE_MS = 10_000');
|
||||
@@ -140,15 +137,15 @@ test('quality judge supervision includes the added judge without changing ordina
|
||||
expect(ALL_TIERS).toEqual({ JUDGE_MS: 120000, CAPTURE_MS: 300000, CAPTURE_LONG_MS: 600000, PTY_MS: 900000, PTY_LONG_MS: 1200000 });
|
||||
const quality = 'test/skill-llm-eval.test.ts';
|
||||
const qualityBudget = FILE_RETRY_BUDGETS.find(row => row.file === quality)!;
|
||||
expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 6_920_000, source: 'registered', policyId: qualityBudget.id });
|
||||
expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 5_960_000, source: 'registered', policyId: qualityBudget.id });
|
||||
expect(retriesForFiles([quality])).toBe(1);
|
||||
const qualitySource = read(quality);
|
||||
const judgeTimeouts = [...qualitySource.matchAll(/}\s*,\s*(JUDGE_MS|WORKFLOW_JUDGE_TEST_MS)\s*\);/g)].map(match => match[1]);
|
||||
expect(judgeTimeouts.filter(timeout => timeout === 'JUDGE_MS')).toHaveLength(11);
|
||||
expect(judgeTimeouts.filter(timeout => timeout === 'JUDGE_MS')).toHaveLength(7);
|
||||
expect(judgeTimeouts.filter(timeout => timeout === 'WORKFLOW_JUDGE_TEST_MS')).toHaveLength(16);
|
||||
expect(qualitySource).toContain('WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000');
|
||||
expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS');
|
||||
expect(qualityBudget.shardMs).toBe((11 * ALL_TIERS.JUDGE_MS + 16 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000);
|
||||
expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 16 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000);
|
||||
expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([
|
||||
[2, 1500000, 1, 6120000], ...Array(5).fill([1, 1500000, 1, 3120000]),
|
||||
]);
|
||||
@@ -213,9 +210,9 @@ test('both gate executors cover the complete census without increasing aggregate
|
||||
expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: slices }, (_, i) => i + 1));
|
||||
expect(planned.slices).toBe(slices);
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceCount: planned.slices, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(58);
|
||||
expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(52);
|
||||
const files = manifest.entries.filter(row => row.status === 'planned').map(row => row.file);
|
||||
expect(new Set(files).size).toBe(58);
|
||||
expect(new Set(files).size).toBe(52);
|
||||
expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected.sort());
|
||||
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
|
||||
manifest.entries.filter(row => row.status === 'planned' && row.slice === slice).map(row => row.file), workers,
|
||||
@@ -250,8 +247,8 @@ test('the periodic executor supervises every actual case and retry within its CI
|
||||
const manifest = buildRunManifest({ tier: 'periodic', sliceCount: planned.slices,
|
||||
dedicatedAutoplanSlice: planned.dedicatedAutoplanSlice, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
const census = manifest.entries.filter(row => row.status === 'planned');
|
||||
expect(census).toHaveLength(100);
|
||||
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_920_000);
|
||||
expect(census).toHaveLength(82);
|
||||
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(5_960_000);
|
||||
expect(manifest.autoplanSlice).toBe(8);
|
||||
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
|
||||
census.filter(row => row.slice === slice).map(row => row.file), active.jobs,
|
||||
|
||||
@@ -60,7 +60,7 @@ describe('paid test enumeration', () => {
|
||||
const files = collectPaidTestFiles();
|
||||
expect(files.length).toBeGreaterThan(0);
|
||||
expect(files.every(isPaidTestFile)).toBe(true);
|
||||
expect(PAID_TEST_GLOBS.length).toBe(7);
|
||||
expect(PAID_TEST_GLOBS.length).toBe(6);
|
||||
|
||||
const shards = planPaidShards(files);
|
||||
expect(shards.flat().sort()).toEqual([...files].sort());
|
||||
|
||||
@@ -93,7 +93,7 @@ describe('periodic fixture dependencies select their behavioral cases', () => {
|
||||
['test/fixtures/ceo-current-record-6aef.json', ['plan-ceo-finding-count']],
|
||||
['test/fixtures/ceo-payment-ledger-decisions.json', ['plan-ceo-finding-count']],
|
||||
['test/setup-gbrain-remote-caller.test.ts', ['setup-gbrain-remote']],
|
||||
['test/skill-fixture.test.ts', ['journey-ideation', 'journey-plan-eng', 'journey-debug', 'journey-qa', 'journey-code-review', 'journey-ship', 'journey-docs', 'journey-retro', 'journey-design-system', 'journey-visual-qa']],
|
||||
['test/skill-fixture.test.ts', ['journey-ideation', 'journey-plan-eng', 'journey-debug', 'journey-qa', 'journey-code-review', 'journey-ship', 'journey-docs', 'journey-retro', 'journey-design-system', 'journey-visual-qa', 'journey-negatives']],
|
||||
['test/office-hours-writeback-env.test.ts', ['office-hours-brain-writeback']],
|
||||
['test/helpers/setup-gbrain-sandbox.ts', ['setup-gbrain-bad-token', 'setup-gbrain-path4-local-pglite', 'setup-gbrain-remote']],
|
||||
['test/helpers/setup-gbrain-fixture-command.ts', ['setup-gbrain-bad-token', 'setup-gbrain-path4-local-pglite']],
|
||||
@@ -154,7 +154,7 @@ describe('periodic fixture dependencies select their behavioral cases', () => {
|
||||
}
|
||||
|
||||
test('SDK runner changes retain the native gate and existing periodic consumers', () => {
|
||||
const periodic = ['brain-privacy-gate', 'setup-gbrain-remote', 'setup-gbrain-bad-token',
|
||||
const periodic = ['setup-gbrain-remote', 'setup-gbrain-bad-token',
|
||||
'setup-gbrain-path4-local-pglite', ...OVERLAY_FIXTURES.map(fixture => `overlay-harness-${fixture.id}`)];
|
||||
const result = selectTests(['test/agent-sdk-runner.test.ts'], E2E_TOUCHFILES);
|
||||
expect(result.reason).toBe('diff');
|
||||
@@ -251,9 +251,9 @@ test('native fixture dependencies include the migrated auto-decision and seeded
|
||||
|
||||
test('shared native input dependencies select every PTY consumer without changing tiers', () => {
|
||||
const expected = selectTests(['test/helpers/claude-pty-runner.ts'], E2E_TOUCHFILES).selected.sort();
|
||||
expect(expected).toHaveLength(22);
|
||||
expect(expected).toHaveLength(20);
|
||||
expect(expected.filter(id => E2E_TIERS[id] === 'gate')).toHaveLength(7);
|
||||
expect(expected.filter(id => E2E_TIERS[id] === 'periodic')).toHaveLength(15);
|
||||
expect(expected.filter(id => E2E_TIERS[id] === 'periodic')).toHaveLength(13);
|
||||
for (const file of ['test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts',
|
||||
'test/helpers/plan-skill-questions.ts', 'test/plan-skill-questions.test.ts', 'test/fixtures/design-tasks-bash-permission.json', 'test/fixtures/eng-auq-validation-error.json',
|
||||
'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts',
|
||||
@@ -266,14 +266,14 @@ test('shared native input dependencies select every PTY consumer without changin
|
||||
|
||||
|
||||
test('seed submission dependencies select every seeded caller with its existing tier', () => {
|
||||
const expected = ['auto-decide-preserved', 'conductor-prose', 'plan-ceo-review-plan-mode',
|
||||
const expected = ['auto-decide-preserved', 'plan-ceo-review-plan-mode',
|
||||
'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-mode-no-op'];
|
||||
for (const file of ['test/helpers/fake-plan-seed.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts']) {
|
||||
const result = selectTests([file], E2E_TOUCHFILES);
|
||||
expect(result.reason).toBe('diff');
|
||||
expect(result.selected.sort()).toEqual(expected);
|
||||
}
|
||||
expect(expected.map(id => E2E_TIERS[id])).toEqual(['periodic', 'periodic', 'gate', 'periodic', 'gate', 'periodic', 'gate']);
|
||||
expect(expected.map(id => E2E_TIERS[id])).toEqual(['periodic', 'gate', 'periodic', 'gate', 'periodic', 'gate']);
|
||||
});
|
||||
|
||||
test('task emission source selects CEO completion consumers', () => {
|
||||
@@ -323,7 +323,6 @@ test('Eng approval-rule source and free contract controls select every declared
|
||||
'plan-review-report',
|
||||
'plan-eng-review-plan-mode',
|
||||
'plan-mode-no-op',
|
||||
'conductor-prose',
|
||||
'carve-section-loading',
|
||||
'autoplan-chain-pty',
|
||||
'plan-eng-finding-count',
|
||||
@@ -334,8 +333,6 @@ test('Eng approval-rule source and free contract controls select every declared
|
||||
'plan-ceo-review-prosons-cadence',
|
||||
'plan-review-prosons-format',
|
||||
'codex-offered-eng-review',
|
||||
'codex-plan-eng-format-coverage',
|
||||
'codex-plan-eng-format-kind',
|
||||
'plan-eng-coverage-audit',
|
||||
'autoplan-dual-voice'
|
||||
];
|
||||
@@ -363,8 +360,7 @@ test('native compact-boundary ancestry selects every consuming callback', () =>
|
||||
'plan-eng-finding-count', 'plan-design-finding-count', 'plan-devex-finding-count',
|
||||
'plan-eng-multi-finding-batching', 'plan-ceo-split-overflow',
|
||||
'plan-design-with-ui-scope', 'plan-design-review-plan-mode', 'plan-eng-review-plan-mode',
|
||||
'auto-decide-preserved', 'conductor-prose',
|
||||
].sort();
|
||||
'auto-decide-preserved', ].sort();
|
||||
for (const file of ['test/helpers/plan-count-transcript.ts', 'test/plan-count-session-cwd.test.ts']) {
|
||||
const selected = selectTests([file], E2E_TOUCHFILES);
|
||||
expect(selected.reason).toBe('diff');
|
||||
@@ -383,7 +379,7 @@ test('same-plan expansion disposition replay selects the existing mode helper co
|
||||
|
||||
|
||||
test('structured auto-decision evidence selects every native observer', () => {
|
||||
const expected = ['auto-decide-preserved', 'conductor-prose', 'plan-ceo-review-plan-mode',
|
||||
const expected = ['auto-decide-preserved', 'plan-ceo-review-plan-mode',
|
||||
'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-mode-no-op'];
|
||||
for (const file of ['test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json',
|
||||
'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json']) {
|
||||
@@ -398,7 +394,7 @@ test('structured auto-decision evidence selects every native observer', () => {
|
||||
|
||||
|
||||
test('explanatory native mode evidence selects all observers with their existing tiers', () => {
|
||||
const expected = ['auto-decide-preserved', 'conductor-prose', 'office-hours-auto-mode',
|
||||
const expected = ['auto-decide-preserved', 'office-hours-auto-mode',
|
||||
'plan-ceo-review-plan-mode', 'plan-design-review-plan-mode', 'plan-devex-review-plan-mode',
|
||||
'plan-eng-review-plan-mode', 'plan-mode-no-op'];
|
||||
for (const file of ['test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts',
|
||||
@@ -408,7 +404,7 @@ test('explanatory native mode evidence selects all observers with their existing
|
||||
expect(selectTests([file], LLM_JUDGE_TOUCHFILES).selected).toEqual([]);
|
||||
}
|
||||
expect(expected.map(id => E2E_TIERS[id])).toEqual([
|
||||
'periodic', 'periodic', 'gate', 'gate', 'periodic', 'gate', 'periodic', 'gate',
|
||||
'periodic', 'gate', 'gate', 'periodic', 'gate', 'periodic', 'gate',
|
||||
]);
|
||||
});
|
||||
|
||||
@@ -432,8 +428,7 @@ test('file supervision regression selects all affected callers with their existi
|
||||
'plan-ceo-review-benefits', 'plan-devex-finding-floor', 'plan-mode-no-op', 'plan-review-report',
|
||||
];
|
||||
const periodic = [
|
||||
'auto-decide-preserved', 'codex-plan-ceo-format-approach', 'codex-plan-ceo-format-mode',
|
||||
'codex-plan-eng-format-coverage', 'codex-plan-eng-format-kind', 'plan-ceo-mode-routing',
|
||||
'auto-decide-preserved', 'plan-ceo-mode-routing',
|
||||
'plan-ceo-review', 'plan-ceo-review-expansion-energy', 'plan-ceo-review-format-approach',
|
||||
'plan-ceo-review-format-mode', 'plan-ceo-review-prosons-cadence', 'plan-ceo-review-selective',
|
||||
'plan-design-finding-floor', 'plan-eng-finding-floor', 'plan-eng-review', 'plan-eng-review-artifact',
|
||||
@@ -498,7 +493,6 @@ const nativeRepairDependencies = [
|
||||
"plan-mode-no-op",
|
||||
"office-hours-auto-mode",
|
||||
"auto-decide-preserved",
|
||||
"conductor-prose"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -595,10 +589,8 @@ const nativeRepairDependencies = [
|
||||
"plan-mode-no-op",
|
||||
"office-hours-auto-mode",
|
||||
"auto-decide-preserved",
|
||||
"conductor-prose",
|
||||
"plan-ceo-mode-routing",
|
||||
"plan-design-with-ui-scope",
|
||||
"ship-idempotency-pty",
|
||||
"autoplan-chain-pty",
|
||||
"plan-ceo-finding-count",
|
||||
"plan-eng-finding-count",
|
||||
@@ -632,7 +624,6 @@ test('native repair dependencies preserve every original tier', () => {
|
||||
"plan-mode-no-op": "gate",
|
||||
"office-hours-auto-mode": "gate",
|
||||
"auto-decide-preserved": "periodic",
|
||||
"conductor-prose": "periodic",
|
||||
"plan-ceo-finding-count": "periodic",
|
||||
"plan-eng-finding-count": "periodic",
|
||||
"plan-design-finding-count": "periodic",
|
||||
@@ -657,7 +648,6 @@ test('promoted public transcript decoder keeps its actual callers selected', ()
|
||||
'plan-eng-review-plan-mode',
|
||||
'plan-design-review-plan-mode',
|
||||
'auto-decide-preserved',
|
||||
'conductor-prose',
|
||||
'plan-ceo-mode-routing',
|
||||
'plan-design-with-ui-scope',
|
||||
'autoplan-chain-pty',
|
||||
@@ -749,7 +739,7 @@ test('numbered native-menu captures select the existing parser consumers', () =>
|
||||
const expected = Object.entries(E2E_TOUCHFILES)
|
||||
.filter(([, files]) => files.includes('test/plan-skill-questions.test.ts'))
|
||||
.map(([id]) => id).sort();
|
||||
expect(expected).toHaveLength(22);
|
||||
expect(expected).toHaveLength(20);
|
||||
for (const file of ['test/pty-numbered-option-indent-native.test.ts',
|
||||
'test/fixtures/ceo-split-e5-numbered-description-491.json']) {
|
||||
expect(selectTests([file], E2E_TOUCHFILES).selected.sort()).toEqual(expected);
|
||||
@@ -808,7 +798,7 @@ test('stderr lifecycle regression selects runtime consumers without a quality-ma
|
||||
'benchmark-workflow', 'setup-deploy-workflow', 'autoplan-dual-voice', 'scrape-match-path', 'scrape-prototype-path',
|
||||
'skillify-happy-path', 'skillify-provenance-refusal', 'skillify-approval-reject', 'journey-ideation', 'journey-plan-eng',
|
||||
'journey-debug', 'journey-qa', 'journey-code-review', 'journey-ship', 'journey-docs',
|
||||
'journey-retro', 'journey-design-system', 'journey-visual-qa', 'fanout-arm-overlay-on', 'fanout-arm-overlay-off',
|
||||
'journey-retro', 'journey-design-system', 'journey-visual-qa', 'journey-negatives',
|
||||
'office-hours-brain-writeback', 'arm-benchmark-native-overbuild', 'arm-benchmark-crud-endpoint', 'arm-benchmark-bugfix-decoys', 'office-hours-section-loading',
|
||||
];
|
||||
const file = 'test/session-runner-stream-lifecycle.test.ts';
|
||||
@@ -831,8 +821,7 @@ for (const file of ['test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures
|
||||
const selected = selectTests([file], E2E_TOUCHFILES);
|
||||
expect(selected.reason).toBe('diff');
|
||||
expect(selected.selected.sort()).toEqual([
|
||||
'auto-decide-preserved', 'autoplan-chain-pty', 'conductor-prose',
|
||||
'plan-ceo-finding-count', 'plan-ceo-mode-routing', 'plan-ceo-split-overflow',
|
||||
'auto-decide-preserved', 'autoplan-chain-pty', 'plan-ceo-finding-count', 'plan-ceo-mode-routing', 'plan-ceo-split-overflow',
|
||||
'plan-design-finding-count', 'plan-design-review-plan-mode', 'plan-design-with-ui-scope',
|
||||
'plan-devex-finding-count', 'plan-eng-finding-count', 'plan-eng-multi-finding-batching',
|
||||
'plan-eng-review-plan-mode',
|
||||
@@ -844,7 +833,7 @@ for (const file of ['test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures
|
||||
|
||||
test('native clipped regressions retain the existing parser and owned-permission selection', () => {
|
||||
for (const [dependency, count, files] of [
|
||||
['test/helpers/claude-pty-runner.ts', 22, [
|
||||
['test/helpers/claude-pty-runner.ts', 20, [
|
||||
'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json',
|
||||
]],
|
||||
['test/helpers/plan-count-file-permission.ts', 10, [
|
||||
|
||||
@@ -11,9 +11,9 @@ const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const source = fs.readFileSync(path.join(ROOT, 'test/skill-e2e-design.test.ts'), 'utf8');
|
||||
const id = 'plan-design-review-plan-mode';
|
||||
|
||||
// Same source-evaluation pattern as plan-tune-cathedral-fixture.test.ts: run
|
||||
// the real suite registration and selected callback, without importing paid
|
||||
// initialization. All filesystem mutations stay in this standalone fixture.
|
||||
// Source-evaluation pattern: run the real suite registration and selected
|
||||
// callback, without importing paid initialization. All filesystem mutations
|
||||
// stay in this standalone fixture.
|
||||
async function exercise(mode: 'success' | 'max-turns' | 'first-timeout' | 'second-timeout' | 'saved-timeout' | 'empty-summary' | 'write-failure' | 'unchanged-seed' | 'no-additions' | 'short-plan' | 'api-error' | 'plan-read-failure' | 'attempt-deadline') {
|
||||
const scratch = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'legacy-design-free-')));
|
||||
const home = path.join(scratch, 'home'); fs.mkdirSync(home);
|
||||
|
||||
@@ -26,12 +26,11 @@ test('semantic helper changes also select the separate DX analysis calibration',
|
||||
// Its direct behavioral consumers extend the unchanged helper-only set.
|
||||
if (file === 'test/plan-review-cases.test.ts') expected.push(
|
||||
'plan-eng-review', 'plan-eng-review-artifact', 'plan-review-report',
|
||||
'plan-eng-review-plan-mode', 'plan-mode-no-op', 'conductor-prose',
|
||||
'plan-eng-review-plan-mode', 'plan-mode-no-op',
|
||||
'carve-section-loading', 'autoplan-chain-pty', 'plan-eng-finding-floor',
|
||||
'plan-eng-review-format-coverage', 'plan-eng-review-format-kind',
|
||||
'plan-ceo-review-prosons-cadence', 'plan-review-prosons-format',
|
||||
'codex-offered-eng-review', 'codex-plan-eng-format-coverage',
|
||||
'codex-plan-eng-format-kind', 'plan-eng-coverage-audit', 'autoplan-dual-voice',
|
||||
'codex-offered-eng-review', 'plan-eng-coverage-audit', 'autoplan-dual-voice',
|
||||
);
|
||||
expect(selectTests([file], E2E_TOUCHFILES, []).selected.sort()).toEqual(
|
||||
expected.sort());
|
||||
|
||||
@@ -1,129 +0,0 @@
|
||||
import { expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { E2E_TOUCHFILES } from './helpers/touchfiles';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-e2e-plan-tune-cathedral.test.ts'), 'utf8');
|
||||
const names = ['plan-tune-hook-capture', 'plan-tune-enforcement', 'plan-tune-annotation', 'plan-tune-codex-import', 'plan-tune-dream-cycle'];
|
||||
|
||||
async function exercise(selected = names, fault?: 'missing-log-lib' | 'missing-hook-lib' | 'first-hook' | 'setup', hostEnv: Record<string, string> = {}) {
|
||||
// Execute the actual selected callbacks with real local bins. Never import
|
||||
// E2E initialization, call a provider, or pass ambient auth/host state.
|
||||
const scratch = fs.mkdtempSync(path.join(os.tmpdir(), 'cathedral-contract-'));
|
||||
const suites: any[] = [], finalizers: any[] = [], rows: any[] = [], attempts: any[] = [], dirs: string[] = [], removals: any[] = [];
|
||||
const invoked: string[] = [];
|
||||
let current: any, failedHook = false;
|
||||
const owned = (p: string) => { if (!p.startsWith(scratch + path.sep)) throw new Error('Foreign fixture path'); };
|
||||
const env = { PATH: process.env.PATH, HOME: path.join(scratch, 'home'), GIT_CONFIG_NOSYSTEM: '1', GIT_CONFIG_GLOBAL: process.platform === 'win32' ? 'NUL' : '/dev/null', ...hostEnv };
|
||||
fs.mkdirSync(env.HOME);
|
||||
const args: Record<string, any> = {
|
||||
ROOT, path, os: {tmpdir:()=>scratch}, process: {env}, expect,
|
||||
beforeAll: (fn: any) => current.setups.push(fn),
|
||||
afterAll: (fn: any) => current ? current.teardowns.push(fn) : finalizers.push(fn),
|
||||
describeIfSelected: (_title: string, keys: string[], fn: any) => {
|
||||
if (!keys.some(key => selected.includes(key))) return;
|
||||
const suite = {setups:[],teardowns:[],tests:[]}; suites.push(suite); current=suite; fn(); current=undefined;
|
||||
},
|
||||
testConcurrentIfSelected: (name: string, fn: any) => { if(selected.includes(name)) current.tests.push({name,fn}); },
|
||||
createEvalCollector: () => ({addTest:(row:any)=>rows.push(row)}), finalizeEvalCollector:()=>{},
|
||||
copyDirSync: (a: string, b: string) => { owned(b); fs.cpSync(a,b,{recursive:true}); },
|
||||
fs: { ...fs,
|
||||
mkdtempSync(prefix: string) { owned(prefix); const dir=fs.mkdtempSync(prefix); dirs.push(dir); return dir; },
|
||||
copyFileSync(a: string,b: string) {
|
||||
owned(b);
|
||||
if(fault==='setup') throw new Error('Synthetic fixture copy failure');
|
||||
if((fault==='missing-log-lib' && a.endsWith('/lib/jsonl-store.ts')) || (fault==='missing-hook-lib' && a.endsWith('/lib/is-conductor.ts'))) return;
|
||||
fs.copyFileSync(a,b);
|
||||
},
|
||||
rmSync(p: string,opts: any) {
|
||||
owned(p); const memory=path.join(p,'.gstack-state/free-text-memory.json');
|
||||
removals.push({path:p,nuggets:fs.existsSync(memory)?JSON.parse(fs.readFileSync(memory,'utf8')).nuggets.length:null});
|
||||
fs.rmSync(p,opts);
|
||||
},
|
||||
},
|
||||
spawnSync: (bin: string,argv: string[],opts: any) => {
|
||||
if(bin!=='git') {
|
||||
owned(bin);
|
||||
expect(['question-log-hook','question-preference-hook','gstack-codex-session-import','gstack-distill-apply']).toContain(path.basename(bin));
|
||||
owned(opts.env.GSTACK_STATE_ROOT);
|
||||
}
|
||||
if(opts.cwd) owned(opts.cwd);
|
||||
expect(Number.isFinite(opts.timeout) && opts.timeout > 0).toBe(true);
|
||||
invoked.push(path.basename(bin));
|
||||
if(fault==='first-hook' && path.basename(bin)==='question-preference-hook' && !failedHook) {
|
||||
failedHook=true; return {status:1,stdout:'',stderr:'Synthetic transient hook failure'};
|
||||
}
|
||||
return spawnSync(bin,argv,{...opts,timeout:opts.timeout,env:opts.env??env});
|
||||
},
|
||||
};
|
||||
let body=source; for(const m of source.matchAll(/^import[\s\S]*?;\n/gm)) body=body.replace(m[0],'');
|
||||
try {
|
||||
new Function(...Object.keys(args),new Bun.Transpiler({loader:'ts'}).transformSync(body))(...Object.values(args));
|
||||
for(const suite of suites) {
|
||||
for(const setup of suite.setups) await setup();
|
||||
for(const callback of suite.tests) for(let attempt=1;attempt<=2;attempt++) {
|
||||
let error: unknown; try {await callback.fn();} catch(e) {error=e;}
|
||||
attempts.push({name:callback.name,attempt,error,remaining:dirs.filter(p=>fs.existsSync(p))});
|
||||
}
|
||||
for(const done of suite.teardowns) await done();
|
||||
}
|
||||
for(const done of finalizers) await done();
|
||||
return {rows,attempts,dirs,removals,invoked,allRemoved:dirs.every(p=>!fs.existsSync(p))};
|
||||
} finally {fs.rmSync(scratch,{recursive:true,force:true});}
|
||||
}
|
||||
|
||||
test('Cathedral callbacks run all five real local contracts with fresh attempts and truthful rows', async () => {
|
||||
const x=await exercise();
|
||||
expect(x.attempts).toHaveLength(10); expect(x.rows).toHaveLength(10); expect(x.dirs).toHaveLength(10);
|
||||
expect(new Set(x.dirs).size).toBe(10); expect(x.allRemoved).toBe(true);
|
||||
for(const attempt of x.attempts) {expect(attempt.error).toBeUndefined();expect(attempt.remaining).toEqual([]);}
|
||||
for(const row of x.rows) {expect(row.passed).toBe(true);expect(row.cost_usd).toBe(0);expect(row.output).toContain('no model invocation');}
|
||||
});
|
||||
|
||||
test('Missing installed libraries still fail the actual hook and log contracts', async () => {
|
||||
for(const [name,fault] of [['plan-tune-hook-capture','missing-log-lib'],['plan-tune-enforcement','missing-hook-lib']] as const) {
|
||||
const x=await exercise([name],fault);
|
||||
expect(x.rows).toHaveLength(2); expect(x.allRemoved).toBe(true);
|
||||
for(const attempt of x.attempts) expect(attempt.error).toBeDefined();
|
||||
expect(x.rows.map(row=>row.passed)).toEqual([false,false]);
|
||||
}
|
||||
});
|
||||
|
||||
test('A failed dream hook retries with one fresh nugget and preserves the failed attempt', async () => {
|
||||
const x=await exercise(['plan-tune-dream-cycle'],'first-hook');
|
||||
expect(x.attempts[0].error).toBeDefined(); expect(x.attempts[1].error).toBeUndefined();
|
||||
expect(x.rows.map(row=>row.passed)).toEqual([false,true]);
|
||||
expect(x.removals.map(row=>row.nuggets)).toEqual([1,1]);
|
||||
expect(new Set(x.dirs).size).toBe(2); expect(x.allRemoved).toBe(true);
|
||||
});
|
||||
|
||||
test('A partial fixture setup failure is recorded once and cleans only its owned directory', async () => {
|
||||
const x=await exercise(['plan-tune-hook-capture'],'setup');
|
||||
expect(x.rows.map(row=>row.passed)).toEqual([false,false]); expect(x.dirs).toHaveLength(2); expect(x.removals).toHaveLength(2);
|
||||
expect(x.invoked.every(bin=>bin==='git')).toBe(true); expect(x.allRemoved).toBe(true);
|
||||
});
|
||||
|
||||
test('Cathedral fixture controls and copied libraries select all five existing owners only', () => {
|
||||
for(const file of ['test/plan-tune-cathedral-fixture.test.ts','lib/jsonl-store.ts','lib/is-conductor.ts']) {
|
||||
const owners=Object.entries(E2E_TOUCHFILES).filter(([name,paths])=>names.includes(name)&&paths.includes(file)).map(([name])=>name);
|
||||
expect(owners).toEqual(names);
|
||||
}
|
||||
expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes('test/plan-tune-cathedral-fixture.test.ts')).map(([name])=>name)).toEqual(names);
|
||||
});
|
||||
|
||||
|
||||
test('Plain Claude cathedral contracts stay isolated from either inherited Conductor marker', async () => {
|
||||
for (const hostEnv of [
|
||||
{ CONDUCTOR_WORKSPACE_PATH: '/synthetic/conductor/workspace' },
|
||||
{ CONDUCTOR_PORT: '55070', GSTACK_SESSION_KIND: 'spawned', OPENCLAW_SESSION: 'synthetic-session' },
|
||||
]) {
|
||||
const x = await exercise(['plan-tune-annotation', 'plan-tune-dream-cycle'], undefined, hostEnv);
|
||||
expect(x.attempts).toHaveLength(4);
|
||||
expect(x.rows.map(row => row.passed)).toEqual([true, true, true, true]);
|
||||
for (const attempt of x.attempts) expect(attempt.error).toBeUndefined();
|
||||
expect(x.allRemoved).toBe(true);
|
||||
}
|
||||
});
|
||||
@@ -1,42 +1,22 @@
|
||||
/**
|
||||
* /plan-tune cathedral E2E (T16) — 5 scenarios, all gate tier per D12.
|
||||
* /plan-tune cathedral contract tests (T16) — 5 scenarios, free suite.
|
||||
*
|
||||
* Each scenario verifies that the cathedral's substrate works end-to-end
|
||||
* through local hook and bin invocations. No model is called: these scenarios
|
||||
* exercise the installed-file contracts, using synthetic hook envelopes and
|
||||
* a synthetic Codex session. Unit tests cover the individual components.
|
||||
*
|
||||
* Touchfile registration in test/helpers/touchfiles.ts:
|
||||
* - plan-tune-hook-capture
|
||||
* - plan-tune-enforcement
|
||||
* - plan-tune-annotation
|
||||
* - plan-tune-codex-import
|
||||
* - plan-tune-dream-cycle
|
||||
*
|
||||
* Each scenario uses GSTACK_STATE_ROOT to isolate from the user's real
|
||||
* ~/.gstack (per cathedral T1 + Codex D16 fix). Every attempt gets a fresh fixture.
|
||||
*/
|
||||
|
||||
import { afterAll, expect } from 'bun:test';
|
||||
import {
|
||||
ROOT,
|
||||
describeIfSelected,
|
||||
testConcurrentIfSelected,
|
||||
copyDirSync,
|
||||
createEvalCollector,
|
||||
finalizeEvalCollector,
|
||||
} from './helpers/e2e-helpers';
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import { ROOT, copyDirSync } from './helpers/e2e-helpers';
|
||||
import { spawnSync } from 'child_process';
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import * as os from 'os';
|
||||
|
||||
const collector = createEvalCollector('e2e-plan-tune-cathedral');
|
||||
|
||||
afterAll(() => {
|
||||
finalizeEvalCollector(collector);
|
||||
});
|
||||
|
||||
/** Scaffold a fixture project with the bins + scripts the cathedral needs. */
|
||||
function scaffoldFixture(workDir: string): { workDir: string; stateRoot: string; slug: string; env: NodeJS.ProcessEnv } {
|
||||
const stateRoot = path.join(workDir, '.gstack-state');
|
||||
@@ -114,27 +94,18 @@ function cleanupFixture(workDir: string): void {
|
||||
}
|
||||
}
|
||||
|
||||
/** Own setup, assertions and cleanup inside each callback, including Bun retries. */
|
||||
/** Own setup, assertions and cleanup inside each callback. */
|
||||
function testCathedral(
|
||||
name: string,
|
||||
prefix: string,
|
||||
run: (fixture: ReturnType<typeof scaffoldFixture>) => Promise<void>,
|
||||
): void {
|
||||
testConcurrentIfSelected(name, async () => {
|
||||
const started = Date.now();
|
||||
let workDir: string | undefined;
|
||||
let passed = false;
|
||||
test(name, async () => {
|
||||
const workDir = fs.mkdtempSync(path.join(os.tmpdir(), prefix));
|
||||
try {
|
||||
workDir = fs.mkdtempSync(path.join(os.tmpdir(), prefix));
|
||||
await run(scaffoldFixture(workDir));
|
||||
passed = true;
|
||||
} finally {
|
||||
if (workDir) cleanupFixture(workDir);
|
||||
collector?.addTest({
|
||||
name, suite: 'plan-tune-cathedral', tier: 'e2e', passed,
|
||||
duration_ms: Date.now() - started, cost_usd: 0,
|
||||
output: 'Local hook/bin contract; no model invocation.',
|
||||
});
|
||||
cleanupFixture(workDir);
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -143,7 +114,7 @@ function testCathedral(
|
||||
// Scenario 1: Hook capture — PostToolUse hook writes to question-log.jsonl
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describeIfSelected('PlanTune cathedral E2E: hook capture', ['plan-tune-hook-capture'], () => {
|
||||
describe('PlanTune cathedral E2E: hook capture', () => {
|
||||
testCathedral('plan-tune-hook-capture', 'cathedral-cap-', async (fixture) => {
|
||||
// Direct hook invocation simulates Claude Code's PostToolUse delivery.
|
||||
// E2E verifies the hook + bin chain works against real bins on disk
|
||||
@@ -190,7 +161,7 @@ describeIfSelected('PlanTune cathedral E2E: hook capture', ['plan-tune-hook-capt
|
||||
// Scenario 2: Enforcement — never-ask preference + marker + 2-way → deny
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describeIfSelected('PlanTune cathedral E2E: enforcement', ['plan-tune-enforcement'], () => {
|
||||
describe('PlanTune cathedral E2E: enforcement', () => {
|
||||
testCathedral('plan-tune-enforcement', 'cathedral-enf-', async (fixture) => {
|
||||
fs.mkdirSync(path.join(fixture.stateRoot, 'projects', fixture.slug), { recursive: true });
|
||||
fs.writeFileSync(
|
||||
@@ -253,7 +224,7 @@ describeIfSelected('PlanTune cathedral E2E: enforcement', ['plan-tune-enforcemen
|
||||
// Scenario 3: Annotation — declared profile injected via additionalContext
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describeIfSelected('PlanTune cathedral E2E: annotation', ['plan-tune-annotation'], () => {
|
||||
describe('PlanTune cathedral E2E: annotation', () => {
|
||||
testCathedral('plan-tune-annotation', 'cathedral-ann-', async (fixture) => {
|
||||
// Strong declared profile that should annotate any signal_key=detail-preference question.
|
||||
fs.writeFileSync(
|
||||
@@ -318,7 +289,7 @@ describeIfSelected('PlanTune cathedral E2E: annotation', ['plan-tune-annotation'
|
||||
// Scenario 4: Codex import — JSONL session → import bin → log fills
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describeIfSelected('PlanTune cathedral E2E: codex import', ['plan-tune-codex-import'], () => {
|
||||
describe('PlanTune cathedral E2E: codex import', () => {
|
||||
testCathedral('plan-tune-codex-import', 'cathedral-cdx-', async (fixture) => {
|
||||
const sessionFile = path.join(fixture.workDir, 'rollout-cathedral.jsonl');
|
||||
const lines = [
|
||||
@@ -374,7 +345,7 @@ describeIfSelected('PlanTune cathedral E2E: codex import', ['plan-tune-codex-imp
|
||||
// re-fire → memory injection
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describeIfSelected('PlanTune cathedral E2E: dream cycle', ['plan-tune-dream-cycle'], () => {
|
||||
describe('PlanTune cathedral E2E: dream cycle', () => {
|
||||
testCathedral('plan-tune-dream-cycle', 'cathedral-dream-', async (fixture) => {
|
||||
// Seed proposals file directly (the SDK call is exercised by the unit
|
||||
// test; here we verify apply → re-fire round-trip on top of a known
|
||||
@@ -87,14 +87,13 @@ describe('session-runner timeout kills the whole process group', () => {
|
||||
}, 60_000);
|
||||
});
|
||||
|
||||
describe('all three provider runners carry the group-kill wiring', () => {
|
||||
// Source pin, not behavior: codex/gemini need their real binaries for a
|
||||
describe('both CLI provider runners carry the group-kill wiring', () => {
|
||||
// Source pin, not behavior: codex needs its real binary for a
|
||||
// behavioral run, but the kill wiring is identical code — a runner that
|
||||
// drops `detached` or reverts to a bare kill() re-opens the orphan class.
|
||||
const runners = [
|
||||
'test/helpers/session-runner.ts',
|
||||
'test/helpers/codex-session-runner.ts',
|
||||
'test/helpers/gemini-session-runner.ts',
|
||||
];
|
||||
for (const rel of runners) {
|
||||
test(`${path.basename(rel)}: detached spawn + killProcessGroup, no bare timeout kill`, () => {
|
||||
|
||||
@@ -1,233 +0,0 @@
|
||||
/**
|
||||
* Privacy-gate E2E (periodic tier, paid).
|
||||
*
|
||||
* The gbrain-sync preamble block instructs the model to fire a one-time
|
||||
* AskUserQuestion when:
|
||||
* - `BRAIN_SYNC: off` in the preamble echo (sync mode not on)
|
||||
* - config `artifacts_sync_mode_prompted` is "false"
|
||||
* - gbrain is detected on the host (binary on PATH or `gbrain doctor`
|
||||
* --fast --json succeeds)
|
||||
*
|
||||
* This test stages all three conditions (via env + a fake `gbrain` binary
|
||||
* on PATH), runs a cheap gstack skill through the Agent SDK, intercepts
|
||||
* every tool use via canUseTool, and asserts: one of the AskUserQuestions
|
||||
* fired by the preamble is the privacy gate with its distinctive prose
|
||||
* and three options (full / artifacts-only / decline).
|
||||
*
|
||||
* Cost: ~$0.30-$0.50 per run. Periodic tier (EVALS=1 EVALS_TIER=periodic).
|
||||
*
|
||||
* See scripts/resolvers/preamble/generate-brain-sync-block.ts for the
|
||||
* prose contract this test locks in.
|
||||
*/
|
||||
|
||||
import { test, expect } from 'bun:test';
|
||||
import { CAPTURE_MS } from './helpers/eval-budgets';
|
||||
import { describeE2ETier } from './helpers/e2e-gate';
|
||||
import * as fs from 'fs';
|
||||
import * as os from 'os';
|
||||
import * as path from 'path';
|
||||
import { runAgentSdkTest, passThroughNonAskUserQuestion, resolveClaudeBinary } from './helpers/agent-sdk-runner';
|
||||
|
||||
const describeE2E = describeE2ETier('periodic');
|
||||
|
||||
describeE2E('gbrain-sync privacy gate fires once via preamble', () => {
|
||||
test('gstack skill preamble fires the 3-option AskUserQuestion when gbrain is detected', async () => {
|
||||
// Stage a fresh GSTACK_HOME with artifacts_sync_mode_prompted=false.
|
||||
const gstackHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-gstack-'));
|
||||
const fakeBinDir = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-bin-'));
|
||||
// Fresh HOME with NO ~/.claude.json: on a machine where gbrain is
|
||||
// registered type=http, the preamble's remote-mode detection reads the
|
||||
// operator's ~/.claude.json and echoes "ARTIFACTS_SYNC: remote-mode" —
|
||||
// and the local privacy gate legitimately never fires. An empty HOME
|
||||
// makes the detection find nothing.
|
||||
const tempHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-home-'));
|
||||
|
||||
// Seed the config so the gate's condition passes.
|
||||
fs.writeFileSync(
|
||||
path.join(gstackHome, 'config.yaml'),
|
||||
'artifacts_sync_mode: off\nartifacts_sync_mode_prompted: false\n',
|
||||
{ mode: 0o600 }
|
||||
);
|
||||
|
||||
// Fake `gbrain` binary that makes the host-detection probe succeed.
|
||||
// The preamble checks `gbrain doctor --fast --json` OR `which gbrain`.
|
||||
// Either branch counts as "gbrain detected."
|
||||
fs.writeFileSync(
|
||||
path.join(fakeBinDir, 'gbrain'),
|
||||
'#!/bin/bash\n' +
|
||||
'case "$1" in\n' +
|
||||
' doctor) echo \'{"status":"ok","schema_version":2}\' ; exit 0 ;;\n' +
|
||||
' --version) echo "0.18.2" ; exit 0 ;;\n' +
|
||||
' *) exit 0 ;;\n' +
|
||||
'esac\n',
|
||||
{ mode: 0o755 }
|
||||
);
|
||||
|
||||
const askUserQuestions: Array<{ input: Record<string, unknown> }> = [];
|
||||
const binary = resolveClaudeBinary();
|
||||
|
||||
// Per-test env, merged LAST by the hermetic env builder (safe post-v1.39:
|
||||
// the runner always passes a COMPLETE hermetic env, so overrides can't
|
||||
// break auth). Ambient process.env.GSTACK_HOME mutation does NOT work
|
||||
// here — hermetic-env scrubs GSTACK_* and repoints GSTACK_HOME at its
|
||||
// own singleton dir, so the staged config would never reach the child.
|
||||
const childEnv = {
|
||||
GSTACK_HOME: gstackHome,
|
||||
HOME: tempHome,
|
||||
PATH: `${fakeBinDir}:${process.env.PATH ?? '/usr/bin:/bin:/opt/homebrew/bin'}`,
|
||||
};
|
||||
|
||||
try {
|
||||
// Pick a small skill with the preamble and load it via Read to force
|
||||
// the model to execute every preamble directive. A narrow "run /learn"
|
||||
// prompt often gets reduced to a direct action, skipping the preamble
|
||||
// gates. Mirror the plan-mode-no-op test pattern: ask the model to
|
||||
// follow the skill's instructions in full.
|
||||
const learnSkill = path.resolve(
|
||||
import.meta.dir,
|
||||
'..',
|
||||
'learn',
|
||||
'SKILL.md'
|
||||
);
|
||||
await runAgentSdkTest({
|
||||
systemPrompt: { type: 'preset', preset: 'claude_code' },
|
||||
userPrompt:
|
||||
`Read the skill file at ${learnSkill} and follow its instructions from the top, including every preamble directive. Execute every bash block. If any AskUserQuestion fires, present it.`,
|
||||
workingDirectory: gstackHome,
|
||||
maxTurns: 10,
|
||||
allowedTools: ['Read', 'Grep', 'Glob', 'Bash'],
|
||||
env: childEnv,
|
||||
...(binary ? { pathToClaudeCodeExecutable: binary } : {}),
|
||||
canUseTool: async (toolName, input) => {
|
||||
if (toolName === 'AskUserQuestion') {
|
||||
askUserQuestions.push({ input });
|
||||
// Auto-answer "Decline — keep everything local" (option C)
|
||||
// so the skill can continue without actually turning on sync.
|
||||
const q = (input.questions as Array<{
|
||||
question: string;
|
||||
options: Array<{ label: string }>;
|
||||
}>)[0];
|
||||
const decline =
|
||||
q.options.find((o) => /decline|keep everything local|no thanks/i.test(o.label)) ??
|
||||
q.options[q.options.length - 1]!;
|
||||
return {
|
||||
behavior: 'allow',
|
||||
updatedInput: {
|
||||
questions: input.questions,
|
||||
answers: { [q.question]: decline.label },
|
||||
},
|
||||
};
|
||||
}
|
||||
return passThroughNonAskUserQuestion(toolName, input);
|
||||
},
|
||||
});
|
||||
|
||||
// Assertion 1: the privacy gate fired.
|
||||
const privacyQuestions = askUserQuestions.filter((aq) => {
|
||||
const qs = aq.input.questions as Array<{ question: string }>;
|
||||
return qs.some(
|
||||
(q) =>
|
||||
/publish.*session memory|private github repo|gbrain indexes/i.test(q.question)
|
||||
);
|
||||
});
|
||||
expect(privacyQuestions.length).toBeGreaterThanOrEqual(1);
|
||||
|
||||
// Assertion 2: the question has the three expected options.
|
||||
const gate = privacyQuestions[0]!.input.questions as Array<{
|
||||
question: string;
|
||||
options: Array<{ label: string }>;
|
||||
}>;
|
||||
const labels = gate[0]!.options.map((o) => o.label.toLowerCase()).join(' | ');
|
||||
// Full / artifacts-only / decline are the three canonical options.
|
||||
expect(labels).toMatch(/everything|allowlisted|full/);
|
||||
expect(labels).toMatch(/artifact/);
|
||||
expect(labels).toMatch(/decline|local|no thanks/);
|
||||
|
||||
// Assertion 3: the gate should NOT fire twice in one run.
|
||||
// (The preamble is supposed to be idempotent within a session.)
|
||||
expect(privacyQuestions.length).toBe(1);
|
||||
} finally {
|
||||
fs.rmSync(gstackHome, { recursive: true, force: true });
|
||||
fs.rmSync(fakeBinDir, { recursive: true, force: true });
|
||||
fs.rmSync(tempHome, { recursive: true, force: true });
|
||||
}
|
||||
}, CAPTURE_MS);
|
||||
|
||||
test('privacy gate does NOT fire when artifacts_sync_mode_prompted is already true', async () => {
|
||||
// Same staging, but prompted=true this time. Gate should be silent.
|
||||
const gstackHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-off-'));
|
||||
const fakeBinDir = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-off-bin-'));
|
||||
// Fresh HOME without a .claude.json — same rationale as the first test:
|
||||
// without it the operator's ~/.claude.json flips the preamble into
|
||||
// remote-mode and this negative test passes vacuously.
|
||||
const tempHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-off-home-'));
|
||||
|
||||
fs.writeFileSync(
|
||||
path.join(gstackHome, 'config.yaml'),
|
||||
'artifacts_sync_mode: off\nartifacts_sync_mode_prompted: true\n',
|
||||
{ mode: 0o600 }
|
||||
);
|
||||
|
||||
fs.writeFileSync(
|
||||
path.join(fakeBinDir, 'gbrain'),
|
||||
'#!/bin/bash\necho \'{"status":"ok"}\'\nexit 0\n',
|
||||
{ mode: 0o755 }
|
||||
);
|
||||
|
||||
const askUserQuestions: Array<{ input: Record<string, unknown> }> = [];
|
||||
const binary = resolveClaudeBinary();
|
||||
|
||||
// Per-test env, merged LAST by the hermetic env builder (see note on the
|
||||
// first test — ambient GSTACK_HOME mutation is scrubbed by hermetic-env).
|
||||
const childEnv = {
|
||||
GSTACK_HOME: gstackHome,
|
||||
HOME: tempHome,
|
||||
PATH: `${fakeBinDir}:${process.env.PATH ?? '/usr/bin:/bin:/opt/homebrew/bin'}`,
|
||||
};
|
||||
|
||||
try {
|
||||
await runAgentSdkTest({
|
||||
systemPrompt: { type: 'preset', preset: 'claude_code' },
|
||||
userPrompt:
|
||||
'Run /learn with no arguments. Just report the learnings count.',
|
||||
workingDirectory: gstackHome,
|
||||
maxTurns: 4,
|
||||
allowedTools: ['Read', 'Grep', 'Glob', 'Bash'],
|
||||
env: childEnv,
|
||||
...(binary ? { pathToClaudeCodeExecutable: binary } : {}),
|
||||
canUseTool: async (toolName, input) => {
|
||||
if (toolName === 'AskUserQuestion') {
|
||||
askUserQuestions.push({ input });
|
||||
// Pass through whatever the model asks; don't prefer anything.
|
||||
const q = (input.questions as Array<{
|
||||
question: string;
|
||||
options: Array<{ label: string }>;
|
||||
}>)[0];
|
||||
return {
|
||||
behavior: 'allow',
|
||||
updatedInput: {
|
||||
questions: input.questions,
|
||||
answers: { [q.question]: q.options[0]!.label },
|
||||
},
|
||||
};
|
||||
}
|
||||
return passThroughNonAskUserQuestion(toolName, input);
|
||||
},
|
||||
});
|
||||
|
||||
// No AskUserQuestion should have matched the privacy gate's prose.
|
||||
const privacyQuestions = askUserQuestions.filter((aq) => {
|
||||
const qs = aq.input.questions as Array<{ question: string }>;
|
||||
return qs.some(
|
||||
(q) =>
|
||||
/publish.*session memory|private github repo|gbrain indexes/i.test(q.question)
|
||||
);
|
||||
});
|
||||
expect(privacyQuestions.length).toBe(0);
|
||||
} finally {
|
||||
fs.rmSync(gstackHome, { recursive: true, force: true });
|
||||
fs.rmSync(fakeBinDir, { recursive: true, force: true });
|
||||
fs.rmSync(tempHome, { recursive: true, force: true });
|
||||
}
|
||||
}, CAPTURE_MS);
|
||||
});
|
||||
@@ -1,76 +0,0 @@
|
||||
/**
|
||||
* Conductor → prose decision brief (periodic-tier, paid, real-PTY).
|
||||
*
|
||||
* Proves the end-to-end behavior: when CONDUCTOR_SESSION is signalled, a skill
|
||||
* that hits a decision renders a PROSE decision brief and waits, instead of
|
||||
* silently skipping the user.
|
||||
*
|
||||
* SCOPE — read before trusting this as the Conductor guard. This is END-TO-END
|
||||
* BEHAVIOR coverage, NOT the discriminating Conductor guarantee:
|
||||
* - The deterministic guard is test/question-preference-hook.test.ts
|
||||
* ("Conductor prose redirect") — it sets process.env.CONDUCTOR_* and asserts
|
||||
* the PreToolUse hook denies + redirects. That test CAN fail on unfixed code.
|
||||
* - The PTY harness here cannot register `mcp__conductor__AskUserQuestion`, so
|
||||
* it tests "native AUQ unavailable + Conductor signal → prose," NOT "the MCP
|
||||
* variant exists and must not be called" (Codex #10). Under --disallowedTools
|
||||
* a present-human interactive session already prose-falls-back, so this test
|
||||
* is a smoke check that the Conductor path still produces a prose brief, not
|
||||
* a proof that the Conductor signal (vs the generic fallback) drove it.
|
||||
*
|
||||
* Periodic tier: model-behavior, non-deterministic.
|
||||
*/
|
||||
|
||||
import { test, expect } from 'bun:test';
|
||||
import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||
import { describeE2ETier } from './helpers/e2e-gate';
|
||||
import { runPlanSkillObservation } from './helpers/claude-pty-runner';
|
||||
|
||||
const describeE2E = describeE2ETier('periodic');
|
||||
|
||||
const FLAWED_PLAN = `# Plan: add a "developer-friendly" pricing tier
|
||||
|
||||
## Goal
|
||||
Increase developer adoption.
|
||||
|
||||
## Premise
|
||||
No tests mentioned, no rollout plan, no auth check on the upgrade endpoint.
|
||||
Adds a Stripe tier, a React pricing page, a Postgres entitlements table, and a
|
||||
Redis cache. The team "feels like" it should be cheaper; no developer was asked.
|
||||
`;
|
||||
|
||||
describeE2E('Conductor renders decisions as prose (periodic)', () => {
|
||||
test('plan-eng-review in a Conductor session surfaces a PROSE decision brief, not a silent skip', async () => {
|
||||
const obs = await runPlanSkillObservation({
|
||||
skillName: 'plan-eng-review',
|
||||
inPlanMode: true,
|
||||
// Mimic Conductor: native AUQ disabled + the Conductor env signal present.
|
||||
extraArgs: ['--disallowedTools', 'AskUserQuestion'],
|
||||
env: { CONDUCTOR_WORKSPACE_PATH: '/tmp/conductor-prose-e2e' },
|
||||
initialPlanContent: FLAWED_PLAN,
|
||||
// A judge can mistake a streaming partial option for a waiting brief.
|
||||
// Keep observing until the independent prose evidence is available.
|
||||
requireProseEvidence: true,
|
||||
timeoutMs: CAPTURE_MS,
|
||||
});
|
||||
|
||||
// The decision must reach the human as prose. 'silent_write' (wrote findings
|
||||
// to the plan without asking) is the precise failure we guard against.
|
||||
if (obs.outcome === 'silent_write') {
|
||||
throw new Error(
|
||||
`Conductor prose regression: skill wrote findings without surfacing a decision.\n` +
|
||||
`summary: ${obs.summary}\n--- evidence ---\n${obs.evidence}`,
|
||||
);
|
||||
}
|
||||
if (obs.outcome === 'exited' || obs.outcome === 'timeout') {
|
||||
throw new Error(
|
||||
`Conductor prose test inconclusive: outcome=${obs.outcome}\n` +
|
||||
`summary: ${obs.summary}\n--- evidence ---\n${obs.evidence}`,
|
||||
);
|
||||
}
|
||||
// A prose-rendered decision brief was observed at some point in the run.
|
||||
expect(obs.proseAUQEverObserved,
|
||||
`Conductor prose decision not observed: outcome=${obs.outcome}\n` +
|
||||
`summary: ${obs.summary}\n--- evidence ---\n${obs.evidence}`,
|
||||
).toBe(true);
|
||||
}, CAPTURE_LONG_MS);
|
||||
});
|
||||
@@ -1,268 +0,0 @@
|
||||
/**
|
||||
* Opus 4.7 behavior evals.
|
||||
*
|
||||
* One case, pinned to claude-opus-4-7:
|
||||
*
|
||||
* Routing precision — the "when in doubt, invoke the skill" policy should
|
||||
* route ambiguous dev prompts to the right skill WITHOUT routing
|
||||
* casual/non-dev prompts. A handful of positive and negative controls.
|
||||
* (The fanout A/B retired 2026-08 — single-run parallel-call comparison was
|
||||
* a coin flip; the SDK overlay-harness is the maintained instrument.)
|
||||
*
|
||||
* Both cases require a running Anthropic API key. Gated behind EVALS=1.
|
||||
* Classify as `periodic` in touchfiles — behavior measurement, not gate.
|
||||
*/
|
||||
|
||||
import { describe, test, expect, afterAll } from 'bun:test';
|
||||
import { JUDGE_MS, CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets';
|
||||
import { runSkillTest } from './helpers/session-runner';
|
||||
import { EvalCollector } from './helpers/eval-store';
|
||||
import { extractSkillHead } from './helpers/skill-fixture';
|
||||
import { spawnSync } from 'child_process';
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import * as os from 'os';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const OPUS_47 = 'claude-opus-4-7';
|
||||
|
||||
const evalsEnabled = !!process.env.EVALS;
|
||||
const describeE2E = evalsEnabled ? describe : describe.skip;
|
||||
const evalCollector = evalsEnabled ? new EvalCollector('e2e-opus-47') : null;
|
||||
const runId = new Date().toISOString().replace(/[:.]/g, '').replace('T', '-').slice(0, 15);
|
||||
|
||||
// --- Helpers ---
|
||||
|
||||
/** Skills that must exist as individual .claude/skills/{name}/SKILL.md files
|
||||
* for Claude Code's auto-discovery to treat them as invokable via Skill tool.
|
||||
* Matches the pattern in skill-routing-e2e.test.ts. */
|
||||
const INSTALLED_SKILLS = [
|
||||
'qa', 'qa-only', 'ship', 'review', 'plan-ceo-review', 'plan-eng-review',
|
||||
'plan-design-review', 'design-review', 'design-consultation', 'retro',
|
||||
'document-release', 'investigate', 'office-hours', 'browse',
|
||||
];
|
||||
|
||||
/** Write a scratch root with:
|
||||
* - Per-skill SKILL.md files under .claude/skills/ (so Skill tool sees them)
|
||||
* - Project CLAUDE.md with explicit routing rules AND (optionally) the
|
||||
* 4.7 overlay content directly inlined so `claude -p` sees it
|
||||
* - git init
|
||||
*
|
||||
* `includeOverlay` controls whether the opus-4-7 nudges (Fan out, Literal,
|
||||
* etc.) get inlined into CLAUDE.md — this is the A/B axis for the fanout
|
||||
* test. `claude -p` doesn't auto-load SKILL.md content, so CLAUDE.md is
|
||||
* the only way to make the overlay visible to the model in this test
|
||||
* harness.
|
||||
*/
|
||||
function mkEvalRoot(suffix: string, includeOverlay: boolean): string {
|
||||
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), `opus47-${suffix}-`));
|
||||
|
||||
// Render at opus-4-7 INTO A MKDTEMP via --out-dir — never the live tree.
|
||||
// The previous cwd=ROOT regeneration rewrote every in-repo SKILL.md
|
||||
// mid-run while concurrent paid shards copyFileSync those same files in
|
||||
// their beforeAll (EVALS_JOBS>=4 locally, 2 per CI slice): a sibling could
|
||||
// capture a half-regenerated or opus-rendered SKILL.md, and a timeout
|
||||
// before afterAll left the whole tree rendered at the wrong model for
|
||||
// every later shard. --out-dir mirrors the repo layout (<out>/<skill>/
|
||||
// SKILL.md), which is all this fixture reads.
|
||||
const renderDir = fs.mkdtempSync(path.join(os.tmpdir(), `opus47-render-${suffix}-`));
|
||||
const result = spawnSync(
|
||||
'bun',
|
||||
['run', 'scripts/gen-skill-docs.ts', '--model', includeOverlay ? 'opus-4-7' : 'claude', '--out-dir', renderDir],
|
||||
{ cwd: ROOT, stdio: 'pipe', encoding: 'utf-8', timeout: 60_000 },
|
||||
);
|
||||
if (result.status !== 0) {
|
||||
throw new Error(`gen-skill-docs failed: ${result.stderr}`);
|
||||
}
|
||||
|
||||
// Install per-skill SKILL.md files for Skill tool discovery. Routing only
|
||||
// reads the frontmatter (name + description), so install frontmatter + the
|
||||
// first ~30 body lines instead of the full 1000-1900-line files
|
||||
// (CLAUDE.md: "E2E test fixtures: extract, don't copy").
|
||||
const skillsDir = path.join(tmp, '.claude', 'skills');
|
||||
for (const skill of INSTALLED_SKILLS) {
|
||||
const src = path.join(renderDir, skill, 'SKILL.md');
|
||||
if (!fs.existsSync(src)) continue;
|
||||
const destDir = path.join(skillsDir, skill);
|
||||
fs.mkdirSync(destDir, { recursive: true });
|
||||
fs.writeFileSync(path.join(destDir, 'SKILL.md'), extractSkillHead(src));
|
||||
}
|
||||
fs.rmSync(renderDir, { recursive: true, force: true });
|
||||
|
||||
// Extract the opus-4-7 model-overlay content from the checked-in file
|
||||
// so we can inline it into CLAUDE.md when includeOverlay is true.
|
||||
const overlayText = includeOverlay
|
||||
? fs.readFileSync(path.join(ROOT, 'model-overlays', 'opus-4-7.md'), 'utf-8')
|
||||
.replace(/\{\{INHERIT:claude\}\}\s*/, '')
|
||||
.trim()
|
||||
: '';
|
||||
|
||||
// Project CLAUDE.md. Explicit routing rules so the agent reaches for
|
||||
// Skill tool on matching prompts, plus the optional overlay.
|
||||
const routingBlock = `## Skill routing
|
||||
|
||||
When the user's request matches an available skill, invoke it via the Skill tool
|
||||
as your FIRST action. The skill has multi-step workflows, checklists, and quality
|
||||
gates that produce better results than an ad-hoc answer. When in doubt, invoke.
|
||||
|
||||
- Bugs, errors, "why is this broken", "wtf" → invoke investigate
|
||||
- Ship, deploy, "send it", create a PR → invoke ship
|
||||
- QA, test the site, "does this work" → invoke qa
|
||||
- Code review, check my diff → invoke review
|
||||
- Product ideas, brainstorming, "is this worth building" → invoke office-hours
|
||||
- Architecture, "does this design make sense" → invoke plan-eng-review
|
||||
- Design system, visual polish → invoke design-review
|
||||
- Weekly retro, what did we ship → invoke retro`;
|
||||
|
||||
const claudeMd = includeOverlay
|
||||
? `# Project\n\n${overlayText}\n\n${routingBlock}\n`
|
||||
: `# Project\n\n${routingBlock}\n`;
|
||||
|
||||
fs.writeFileSync(path.join(tmp, 'CLAUDE.md'), claudeMd);
|
||||
fs.writeFileSync(path.join(tmp, 'package.json'), '{"name":"opus47-eval"}');
|
||||
|
||||
const git = (args: string[]) =>
|
||||
spawnSync('git', args, { cwd: tmp, stdio: 'pipe', timeout: 5_000 });
|
||||
git(['init']);
|
||||
git(['config', 'user.email', 't@t.com']);
|
||||
git(['config', 'user.name', 'T']);
|
||||
git(['add', '.']);
|
||||
git(['commit', '-m', 'init']);
|
||||
|
||||
return tmp;
|
||||
}
|
||||
|
||||
/** Count parallel tool calls in the first assistant turn. */
|
||||
function firstTurnParallelism(transcript: any[]): number {
|
||||
const firstAssistant = transcript.find((e) => e.type === 'assistant');
|
||||
if (!firstAssistant) return 0;
|
||||
const content = firstAssistant.message?.content ?? [];
|
||||
return content.filter((c: any) => c.type === 'tool_use').length;
|
||||
}
|
||||
|
||||
interface RoutingCase {
|
||||
name: string;
|
||||
prompt: string;
|
||||
shouldRoute: boolean;
|
||||
expectedSkill?: string;
|
||||
}
|
||||
|
||||
/** Small, intentionally chosen routing cases. Positive cases are ambiguous
|
||||
* phrasings the user actually says, not template text. Negative cases are
|
||||
* casual or off-topic prompts that match routing keywords but shouldn't
|
||||
* trigger a skill. */
|
||||
const ROUTING_CASES: RoutingCase[] = [
|
||||
// Positive — should route
|
||||
{ name: 'pos-wtf-bug', prompt: "wtf is this error coming from auth.ts:47 when the cookie expires?", shouldRoute: true, expectedSkill: 'investigate' },
|
||||
{ name: 'pos-send-it', prompt: "ok this is good enough, let's send it.", shouldRoute: true, expectedSkill: 'ship' },
|
||||
{ name: 'pos-does-it-work', prompt: "I just pushed the login flow changes. Test the deployed site and find any bugs.", shouldRoute: true, expectedSkill: 'qa' },
|
||||
// Negative — should NOT route
|
||||
{ name: 'neg-syntax-q', prompt: "wtf does this Python list comprehension syntax even mean, [x for x in y if z]?", shouldRoute: false },
|
||||
{ name: 'neg-algo-q', prompt: "does this bubble sort algorithm actually work in O(n log n)?", shouldRoute: false },
|
||||
{ name: 'neg-slack-send', prompt: "can you help me write the slack message? I want to send it to the team.", shouldRoute: false },
|
||||
];
|
||||
|
||||
// --- Tests ---
|
||||
|
||||
describeE2E('Opus 4.7 overlay behavior evals', () => {
|
||||
afterAll(() => {
|
||||
evalCollector?.finalize();
|
||||
// No tree restore needed: mkEvalRoot renders into a mkdtemp via
|
||||
// --out-dir, so the live repo's SKILL.md files are never touched — a
|
||||
// timeout mid-run can no longer strand the tree at the wrong model for
|
||||
// concurrent shards.
|
||||
});
|
||||
|
||||
// (fanout A/B retired, 2026-08 audit: it compared parallel-call counts of
|
||||
// two SINGLE stochastic runs — parA >= parB is a coin flip with near-zero
|
||||
// remaining information; the overlay-fanout question is answered and the
|
||||
// SDK overlay-harness (test/skill-e2e-overlay-harness.test.ts) is the
|
||||
// maintained instrument for the next experiment.)
|
||||
|
||||
|
||||
test(
|
||||
'routing precision: positives route, negatives do not',
|
||||
async () => {
|
||||
// Single SKILL.md tree shared by all cases. We run claude-opus-4-7 with
|
||||
// tool access to Skill; measure whether the first tool call is Skill(..)
|
||||
// and if so, which skill.
|
||||
const root = mkEvalRoot('routing', true);
|
||||
|
||||
try {
|
||||
const results = await Promise.all(
|
||||
ROUTING_CASES.map((c) =>
|
||||
runSkillTest({
|
||||
prompt: c.prompt,
|
||||
workingDirectory: root,
|
||||
maxTurns: 3,
|
||||
allowedTools: ['Skill', 'Read', 'Bash', 'Glob', 'Grep'],
|
||||
timeout: JUDGE_MS,
|
||||
testName: `routing-${c.name}`,
|
||||
runId,
|
||||
model: OPUS_47,
|
||||
}).then((r) => ({ c, r })),
|
||||
),
|
||||
);
|
||||
|
||||
let tp = 0, fn = 0, fp = 0, tn = 0;
|
||||
const rows: string[] = [];
|
||||
let totalCost = 0;
|
||||
|
||||
for (const { c, r } of results) {
|
||||
const skillCalls = r.toolCalls.filter((tc) => tc.tool === 'Skill');
|
||||
const routed = skillCalls.length > 0;
|
||||
const actualSkill = routed ? skillCalls[0]?.input?.skill : undefined;
|
||||
|
||||
const correct = c.shouldRoute
|
||||
? routed && (!c.expectedSkill || actualSkill === c.expectedSkill)
|
||||
: !routed;
|
||||
|
||||
if (c.shouldRoute && routed) tp++;
|
||||
else if (c.shouldRoute && !routed) fn++;
|
||||
else if (!c.shouldRoute && routed) fp++;
|
||||
else tn++;
|
||||
|
||||
totalCost += r.costEstimate.estimatedCost;
|
||||
rows.push(
|
||||
` ${c.name.padEnd(18)} routed=${String(routed).padEnd(5)} skill=${String(actualSkill).padEnd(16)} ` +
|
||||
`expected=${c.shouldRoute ? (c.expectedSkill ?? 'any') : '(none)'} ${correct ? 'OK' : 'MISS'}`,
|
||||
);
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: `routing-${c.name}`,
|
||||
suite: 'Opus 4.7 routing',
|
||||
tier: 'e2e',
|
||||
passed: correct,
|
||||
duration_ms: r.duration,
|
||||
cost_usd: r.costEstimate.estimatedCost,
|
||||
transcript: r.transcript,
|
||||
output: `routed=${routed} actual=${actualSkill ?? '(none)'} expected=${c.shouldRoute ? c.expectedSkill ?? 'any' : '(none)'}`,
|
||||
turns_used: r.costEstimate.turnsUsed,
|
||||
exit_reason: r.exitReason,
|
||||
});
|
||||
}
|
||||
|
||||
const posCount = ROUTING_CASES.filter((c) => c.shouldRoute).length;
|
||||
const negCount = ROUTING_CASES.length - posCount;
|
||||
const tpRate = posCount > 0 ? tp / posCount : 0;
|
||||
const fpRate = negCount > 0 ? fp / negCount : 0;
|
||||
|
||||
console.log(`[opus-4-7 routing] total cost $${totalCost.toFixed(2)}`);
|
||||
console.log(rows.join('\n'));
|
||||
console.log(
|
||||
` TP=${tp}/${posCount} (${(tpRate * 100).toFixed(0)}%) FN=${fn} ` +
|
||||
`FP=${fp}/${negCount} (${(fpRate * 100).toFixed(0)}%) TN=${tn}`,
|
||||
);
|
||||
|
||||
// Thresholds from the test plan artifact: TP >= 80%, FP <= 30%.
|
||||
// With a small N we loosen slightly: TP >= 66% (2 of 3 positive),
|
||||
// FP <= 33% (no more than 1 of 3 negatives).
|
||||
expect(tpRate, `true-positive rate ${(tpRate * 100).toFixed(0)}% (need >= 66%)`).toBeGreaterThanOrEqual(2 / 3);
|
||||
expect(fpRate, `false-positive rate ${(fpRate * 100).toFixed(0)}% (need <= 33%)`).toBeLessThanOrEqual(1 / 3);
|
||||
} finally {
|
||||
fs.rmSync(root, { recursive: true, force: true });
|
||||
}
|
||||
},
|
||||
CAPTURE_LONG_MS,
|
||||
);
|
||||
});
|
||||
@@ -1,6 +0,0 @@
|
||||
import { describeE2ETier } from './helpers/e2e-gate';
|
||||
import { registerOverlayCase } from './helpers/overlay-case';
|
||||
|
||||
describeE2ETier('periodic')('overlay behavior contract v2 (SDK)', () => {
|
||||
registerOverlayCase('opus-4-7-effort-match-trivial-sonnet');
|
||||
});
|
||||
@@ -1,6 +0,0 @@
|
||||
import { describeE2ETier } from './helpers/e2e-gate';
|
||||
import { registerOverlayCase } from './helpers/overlay-case';
|
||||
|
||||
describeE2ETier('periodic')('overlay behavior contract v2 (SDK)', () => {
|
||||
registerOverlayCase('opus-4-7-literal-interpretation-sonnet');
|
||||
});
|
||||
@@ -1,285 +0,0 @@
|
||||
/**
|
||||
* /ship idempotency E2E (periodic, paid, real-PTY).
|
||||
*
|
||||
* Asserts: when /ship runs against a branch that has ALREADY been bumped
|
||||
* (VERSION ahead of base AND package.json synced AND a CHANGELOG entry
|
||||
* exists for the bumped version), the workflow:
|
||||
*
|
||||
* 1. Detects ALREADY_BUMPED state via the Step 12 idempotency check
|
||||
* 2. Does NOT echo STATE: FRESH (which would trigger a second bump)
|
||||
* 3. Does NOT mutate the fixture's VERSION file
|
||||
* 4. Does NOT append a duplicate CHANGELOG [0.0.2] entry
|
||||
* 5. Does NOT create a new "chore: bump version" commit
|
||||
*
|
||||
* Why real-PTY: the old SDK-harness ship-idempotency variant (removed in
|
||||
* v1.64.1.0 as redundant with this test) used a synthetic prompt asking
|
||||
* the agent to "run ONLY the idempotency checks." This test exercises the
|
||||
* actual /ship skill end-to-end against a real git fixture so a regression
|
||||
* that silently re-bumps despite the check passing would be caught.
|
||||
*
|
||||
* Plan-mode framing: we run /ship in plan mode so the agent cannot push,
|
||||
* commit, or open PRs. The Step 12 idempotency check is read-only
|
||||
* (reads VERSION + package.json + git rev-parse) and runs fine in plan
|
||||
* mode. The plan-ready output serves as the terminal signal — the agent
|
||||
* has done its analysis and produced a plan describing what it would do.
|
||||
*
|
||||
* If the agent decides to bump or push despite the fixture's
|
||||
* ALREADY_BUMPED state, that intent surfaces in the plan or in
|
||||
* tool-call attempts, which we detect.
|
||||
*
|
||||
* Cost: ~$2-4/run. Periodic tier — long, runs weekly.
|
||||
*/
|
||||
|
||||
import { test, expect } from 'bun:test';
|
||||
import { PTY_LONG_MS } from './helpers/eval-budgets';
|
||||
import { describeE2ETier } from './helpers/e2e-gate';
|
||||
import { spawnSync } from 'child_process';
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import * as os from 'os';
|
||||
import {
|
||||
launchClaudePty,
|
||||
isPermissionDialogVisible,
|
||||
isNumberedOptionListVisible,
|
||||
} from './helpers/claude-pty-runner';
|
||||
|
||||
const describeE2E = describeE2ETier('periodic');
|
||||
|
||||
interface ShipFixture {
|
||||
workTree: string;
|
||||
bareRemote: string;
|
||||
/** Full bash log of `git` and helper commands run during setup. */
|
||||
setupLog: string[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a self-contained git fixture representing an already-shipped state:
|
||||
* - main branch at VERSION 0.0.1, with one CHANGELOG entry [0.0.1]
|
||||
* - feat/already-shipped branch at VERSION 0.0.2 (bumped + synced),
|
||||
* CHANGELOG has [0.0.2] entry on top of [0.0.1], one feature commit
|
||||
* - bareRemote is the origin; both branches are pushed
|
||||
*
|
||||
* Returns the work-tree dir for /ship to operate on.
|
||||
*/
|
||||
function buildShippedFixture(): ShipFixture {
|
||||
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ship-fixture-'));
|
||||
const workTree = path.join(root, 'workspace');
|
||||
const bareRemote = path.join(root, 'origin.git');
|
||||
fs.mkdirSync(workTree, { recursive: true });
|
||||
|
||||
const setupLog: string[] = [];
|
||||
const sh = (cmd: string, cwd: string): void => {
|
||||
setupLog.push(`[${cwd}] ${cmd}`);
|
||||
const result = spawnSync('bash', ['-c', cmd], { cwd, stdio: 'pipe', timeout: 15_000 });
|
||||
if (result.status !== 0) {
|
||||
const stderr = result.stderr?.toString() ?? '';
|
||||
throw new Error(`fixture setup failed at "${cmd}":\n${stderr}\n--- log ---\n${setupLog.join('\n')}`);
|
||||
}
|
||||
};
|
||||
|
||||
// Bare remote.
|
||||
sh(`git init --bare "${bareRemote}"`, root);
|
||||
|
||||
// Initial commit on main.
|
||||
sh('git init -b main', workTree);
|
||||
sh('git config user.email "test@test.com"', workTree);
|
||||
sh('git config user.name "Test"', workTree);
|
||||
sh('git config commit.gpgsign false', workTree);
|
||||
|
||||
fs.writeFileSync(path.join(workTree, 'VERSION'), '0.0.1\n');
|
||||
fs.writeFileSync(
|
||||
path.join(workTree, 'package.json'),
|
||||
JSON.stringify({ name: 'fixture', version: '0.0.1', private: true }, null, 2) + '\n',
|
||||
);
|
||||
fs.writeFileSync(
|
||||
path.join(workTree, 'CHANGELOG.md'),
|
||||
`# Changelog\n\n## [0.0.1] - 2026-01-01\n\n- Initial release\n`,
|
||||
);
|
||||
fs.writeFileSync(path.join(workTree, 'README.md'), '# Fixture\n');
|
||||
|
||||
sh('git add VERSION package.json CHANGELOG.md README.md', workTree);
|
||||
sh('git commit -m "chore: initial release v0.0.1"', workTree);
|
||||
sh(`git remote add origin "${bareRemote}"`, workTree);
|
||||
sh('git push -u origin main', workTree);
|
||||
|
||||
// Feature branch with ALREADY_BUMPED state.
|
||||
sh('git checkout -b feat/already-shipped', workTree);
|
||||
fs.writeFileSync(path.join(workTree, 'VERSION'), '0.0.2\n');
|
||||
fs.writeFileSync(
|
||||
path.join(workTree, 'package.json'),
|
||||
JSON.stringify({ name: 'fixture', version: '0.0.2', private: true }, null, 2) + '\n',
|
||||
);
|
||||
fs.writeFileSync(
|
||||
path.join(workTree, 'CHANGELOG.md'),
|
||||
`# Changelog\n\n## [0.0.2] - 2026-04-25\n\n**Feature shipped.**\n\nAdded the new feature.\n\n## [0.0.1] - 2026-01-01\n\n- Initial release\n`,
|
||||
);
|
||||
fs.writeFileSync(path.join(workTree, 'feature.md'), '# Feature\n\nAlready shipped.\n');
|
||||
|
||||
sh('git add VERSION package.json CHANGELOG.md feature.md', workTree);
|
||||
sh('git commit -m "feat: add new feature\n\nbumps VERSION to 0.0.2"', workTree);
|
||||
sh('git push -u origin feat/already-shipped', workTree);
|
||||
|
||||
return { workTree, bareRemote, setupLog };
|
||||
}
|
||||
|
||||
/** Snapshot the load-bearing fixture state so we can compare post-run. */
|
||||
interface FixtureSnapshot {
|
||||
versionFile: string;
|
||||
packageVersion: string;
|
||||
changelogEntryCount: number;
|
||||
bumpCommitCount: number;
|
||||
branchHead: string;
|
||||
}
|
||||
|
||||
function snapshotFixture(workTree: string): FixtureSnapshot {
|
||||
const versionFile = fs.readFileSync(path.join(workTree, 'VERSION'), 'utf-8').trim();
|
||||
const pkg = JSON.parse(fs.readFileSync(path.join(workTree, 'package.json'), 'utf-8'));
|
||||
const changelog = fs.readFileSync(path.join(workTree, 'CHANGELOG.md'), 'utf-8');
|
||||
// Count `## [0.0.2]` headings — should stay at 1 across re-runs.
|
||||
const changelogEntryCount = (changelog.match(/^##\s*\[0\.0\.2\]/gm) ?? []).length;
|
||||
const head = spawnSync('git', ['rev-parse', 'HEAD'], { cwd: workTree, stdio: 'pipe', timeout: 30_000 });
|
||||
const branchHead = head.stdout?.toString().trim() ?? '';
|
||||
// Count "chore: bump version" commits on this branch since main.
|
||||
const log = spawnSync(
|
||||
'git', ['log', '--format=%s', 'main..HEAD'],
|
||||
{ cwd: workTree, stdio: 'pipe', timeout: 30_000 },
|
||||
);
|
||||
const subjects = log.stdout?.toString() ?? '';
|
||||
const bumpCommitCount = subjects.split('\n').filter(s => /chore:\s*bump\s+version/i.test(s)).length;
|
||||
return { versionFile, packageVersion: pkg.version, changelogEntryCount, bumpCommitCount, branchHead };
|
||||
}
|
||||
|
||||
describeE2E('/ship idempotency E2E (periodic, real-PTY)', () => {
|
||||
test(
|
||||
'rerunning /ship on an already-shipped branch detects ALREADY_BUMPED and does not mutate fixture',
|
||||
async () => {
|
||||
const fixture = buildShippedFixture();
|
||||
const before = snapshotFixture(fixture.workTree);
|
||||
|
||||
const session = await launchClaudePty({
|
||||
permissionMode: 'plan',
|
||||
cwd: fixture.workTree,
|
||||
timeoutMs: PTY_LONG_MS,
|
||||
// Disable network-y pieces so the agent can't reach actual github.
|
||||
env: { GH_TOKEN: 'mock-not-real', NO_COLOR: '1' },
|
||||
seedSkills: true,
|
||||
});
|
||||
|
||||
let outcome: 'detected' | 'plan_ready' | 'attempted_mutation' | 'timeout' | 'exited' = 'timeout';
|
||||
let evidence = '';
|
||||
|
||||
try {
|
||||
await Bun.sleep(8000);
|
||||
const since = session.mark();
|
||||
session.send('/ship\r');
|
||||
|
||||
const budgetMs = 900_000;
|
||||
const start = Date.now();
|
||||
let lastPermSig = '';
|
||||
while (Date.now() - start < budgetMs) {
|
||||
await Bun.sleep(3000);
|
||||
if (session.exited()) {
|
||||
outcome = 'exited';
|
||||
evidence = session.visibleSince(since).slice(-3000);
|
||||
break;
|
||||
}
|
||||
const visible = session.visibleSince(since);
|
||||
|
||||
// Auto-grant any permission dialogs the preamble triggers
|
||||
// (e.g. touch on a marker file claude considers sensitive).
|
||||
// Classify on the recent tail; don't double-press the same render.
|
||||
const tail = visible.slice(-1500);
|
||||
if (isNumberedOptionListVisible(tail) && isPermissionDialogVisible(tail)) {
|
||||
const sig = visible.slice(-500);
|
||||
if (sig !== lastPermSig) {
|
||||
lastPermSig = sig;
|
||||
session.send('1\r');
|
||||
await Bun.sleep(1500);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Positive: idempotency classify reported ALREADY_BUMPED. Post-carve
|
||||
// (T9), Step 12 runs `gstack-version-bump classify` which emits JSON
|
||||
// (`"state":"ALREADY_BUMPED"`); the legacy inline bash echoed
|
||||
// `STATE: ALREADY_BUMPED`. Accept either so the test survives the carve.
|
||||
if (/STATE:\s*ALREADY_BUMPED|"state":\s*"ALREADY_BUMPED"/.test(visible)) {
|
||||
outcome = 'detected';
|
||||
evidence = visible.slice(-3000);
|
||||
break;
|
||||
}
|
||||
|
||||
// Negative regressions:
|
||||
// - classify reported FRESH (CLI JSON or legacy echo) → would re-bump
|
||||
// - agent attempted git commit -m "chore: bump version"
|
||||
// - agent attempted git push
|
||||
// - agent ran the CLI write path (gstack-version-bump write) — a
|
||||
// re-bump on an already-shipped branch
|
||||
if (
|
||||
/"state":\s*"FRESH"/.test(visible) ||
|
||||
/STATE:\s*FRESH(?![\w-])/i.test(visible) ||
|
||||
/gstack-version-bump\s+write/i.test(visible) ||
|
||||
/git\s+commit\s+.*chore:\s*bump\s+version/i.test(visible) ||
|
||||
/git\s+push.*origin/i.test(visible)
|
||||
) {
|
||||
outcome = 'attempted_mutation';
|
||||
evidence = visible.slice(-3000);
|
||||
break;
|
||||
}
|
||||
|
||||
// Plan-ready outcome (acceptable terminal): the agent finished
|
||||
// analysis. We'll accept this if no mutation signals showed up.
|
||||
if (/ready to execute|Would you like to proceed/i.test(visible)) {
|
||||
outcome = 'plan_ready';
|
||||
evidence = visible.slice(-3000);
|
||||
break;
|
||||
}
|
||||
}
|
||||
// Budget exhausted without a terminal signal: capture the tail NOW,
|
||||
// while the session is still alive. Only the break paths above set
|
||||
// evidence — without this, the timeout throw ships evidence: "".
|
||||
if (outcome === 'timeout') {
|
||||
evidence = session.visibleSince(since).slice(-3000);
|
||||
}
|
||||
} finally {
|
||||
await session.close();
|
||||
}
|
||||
|
||||
// Verify fixture was not mutated regardless of outcome.
|
||||
const after = snapshotFixture(fixture.workTree);
|
||||
const fixtureStable =
|
||||
after.versionFile === before.versionFile &&
|
||||
after.packageVersion === before.packageVersion &&
|
||||
after.changelogEntryCount === before.changelogEntryCount &&
|
||||
after.bumpCommitCount === before.bumpCommitCount &&
|
||||
after.branchHead === before.branchHead;
|
||||
|
||||
try {
|
||||
if (outcome === 'attempted_mutation') {
|
||||
throw new Error(
|
||||
`/ship attempted to mutate already-shipped state.\n` +
|
||||
`--- evidence (last 3KB) ---\n${evidence}\n` +
|
||||
`--- before ---\n${JSON.stringify(before, null, 2)}\n` +
|
||||
`--- after ---\n${JSON.stringify(after, null, 2)}`,
|
||||
);
|
||||
}
|
||||
if (outcome === 'exited') {
|
||||
throw new Error(`claude exited unexpectedly.\n--- evidence ---\n${evidence}`);
|
||||
}
|
||||
if (outcome === 'timeout') {
|
||||
throw new Error(
|
||||
`Timed out before any terminal outcome.\n--- evidence (last 3KB) ---\n${evidence}`,
|
||||
);
|
||||
}
|
||||
// Detected or plan_ready — both are acceptable terminal outcomes.
|
||||
expect(['detected', 'plan_ready']).toContain(outcome);
|
||||
// Fixture must not have been mutated regardless of outcome.
|
||||
expect(fixtureStable).toBe(true);
|
||||
} finally {
|
||||
// Clean up fixture root.
|
||||
try { fs.rmSync(path.dirname(fixture.workTree), { recursive: true, force: true }); } catch { /* ignore */ }
|
||||
}
|
||||
},
|
||||
PTY_LONG_MS, // 20 min wall clock
|
||||
);
|
||||
});
|
||||
@@ -1,34 +0,0 @@
|
||||
/**
|
||||
* /spec --execute end-to-end (periodic, paid, real-PTY).
|
||||
*
|
||||
* Asserts: when /spec --execute runs against a fixture prompt, it:
|
||||
* 1. Refuses to draft on turn 1 (Phase 1 hard gate)
|
||||
* 2. Reads code in Phase 3 (cites a real file path from the fixture repo)
|
||||
* 3. Passes the quality gate (score >= 7) on a well-formed fixture
|
||||
* 4. Spawns a fresh worktree on branch spec/<slug>-<pid>
|
||||
* 5. Issues a final-confirm AskUserQuestion before the spawn
|
||||
*
|
||||
* Cost: ~$3-5/run, 5-8 min wall clock. Periodic — runs weekly via cron or
|
||||
* on demand via `EVALS=1 EVALS_TIER=periodic bun run test:e2e`.
|
||||
*
|
||||
* TODO (v1.1): expand to test all 5 expansion paths and the plan-mode-aware
|
||||
* Phase 5 branching (active vs inactive). Current implementation is the
|
||||
* minimum smoke that proves --execute end-to-end works.
|
||||
*/
|
||||
|
||||
import { test } from 'bun:test';
|
||||
import { describeE2ETier } from './helpers/e2e-gate';
|
||||
|
||||
const describeE2E = describeE2ETier('periodic');
|
||||
|
||||
describeE2E('/spec --execute end-to-end (periodic)', () => {
|
||||
// test.todo, not expect(true): the placeholder reported PASS on every
|
||||
// periodic run while asserting nothing — a lying green with a 600s budget.
|
||||
// The file itself stays: it is the periodic-tier surface registered in
|
||||
// E2E_TIERS so the diff-based selector runs it when spec/ changes, and
|
||||
// the deterministic template-invariant coverage in
|
||||
// spec-template-invariants.test.ts + spec-template-sync.test.ts gates the
|
||||
// gate tier. Implementation spec for the real PTY-driven test lives in
|
||||
// the header TODO ("/spec --execute E2E full pipeline test (v1.1)").
|
||||
test.todo('phase gating + magical Phase 3 + quality gate + spawn — full pipeline');
|
||||
});
|
||||
@@ -1,35 +0,0 @@
|
||||
/**
|
||||
* /spec LLM-judge eval (periodic, paid).
|
||||
*
|
||||
* Asserts: when /spec runs against a fixture vague request, the agent
|
||||
* produces a spec body that scores >= 8/10 against an LLM judge using
|
||||
* the contributor's 14 Quality Standards as the rubric.
|
||||
*
|
||||
* Cost: ~$0.15/run. Periodic — runs weekly via cron or on demand via
|
||||
* `EVALS=1 EVALS_TIER=periodic bun run test:evals`.
|
||||
*
|
||||
* TODO (v1.1): expand fixture set to cover bug / feature / refactor / audit
|
||||
* framings + project-level prompts (no concrete file mapping, exercises the
|
||||
* Phase 3 fallback path).
|
||||
*/
|
||||
|
||||
import { describe, test } from 'bun:test';
|
||||
|
||||
const evalsEnabled = !!process.env.EVALS;
|
||||
const describeEval = evalsEnabled ? describe : describe.skip;
|
||||
|
||||
describeEval('/spec LLM-judge eval (periodic)', () => {
|
||||
// test.todo, not expect(true): the placeholder reported PASS on every
|
||||
// run while asserting nothing — a lying green with a 300s budget. The
|
||||
// file stays as the periodic-tier selector surface for spec/ changes.
|
||||
//
|
||||
// Expected v1.1 implementation:
|
||||
// 1. Pick fixture prompt from test/fixtures/spec/vague-bug.md
|
||||
// 2. Spawn `claude -p` with /spec loaded, send the prompt + role-play
|
||||
// five Phase 1 answers (from test/fixtures/spec/vague-bug-answers.json)
|
||||
// 3. Capture final spec body
|
||||
// 4. Dispatch to Claude judge with prompt encoding the 14 Quality
|
||||
// Standards from spec/SKILL.md.tmpl
|
||||
// 5. Assert numeric score >= 8
|
||||
test.todo('spec body scores >= 8/10 against 14-standard rubric on fixture request');
|
||||
});
|
||||
+21
-199
@@ -12,7 +12,6 @@
|
||||
|
||||
import { afterAll, expect } from 'bun:test';
|
||||
import { JUDGE_MS } from './helpers/eval-budgets';
|
||||
import Anthropic from '@anthropic-ai/sdk';
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge';
|
||||
@@ -62,12 +61,11 @@ function readBrowseCommandSection(): string {
|
||||
}
|
||||
|
||||
/** Slice a section out of the command-list section file, guarded non-empty. */
|
||||
function sliceBrowseSection(startHeader: string, endHeader?: string): string {
|
||||
function sliceBrowseSection(startHeader: string): string {
|
||||
const content = readBrowseCommandSection();
|
||||
const start = content.indexOf(startHeader);
|
||||
if (start < 0) throw new Error(`browse/sections/command-list.md: "${startHeader}" not found`);
|
||||
const end = endHeader ? content.indexOf(endHeader) : -1;
|
||||
const section = end > start ? content.slice(start, end) : content.slice(start);
|
||||
const section = content.slice(start);
|
||||
if (section.trim().length < 200) {
|
||||
throw new Error(`browse/sections/command-list.md slice at "${startHeader}" is empty/stub — regenerate with: bun run gen:skill-docs`);
|
||||
}
|
||||
@@ -91,85 +89,45 @@ function testIfSelected(testName: string, fn: () => Promise<void>, timeout: numb
|
||||
}
|
||||
|
||||
describeIfSelected('LLM-as-judge quality evals', [
|
||||
'command reference table', 'snapshot flags reference',
|
||||
'browse/SKILL.md reference', 'setup block', 'regression vs baseline',
|
||||
'browse/SKILL.md reference', 'setup block',
|
||||
], () => {
|
||||
testIfSelected('command reference table', async () => {
|
||||
const t0 = Date.now();
|
||||
// Browse carve: the command reference lives in the generated on-demand
|
||||
// section browse/sections/command-list.md now (read via non-empty guard).
|
||||
const section = sliceBrowseSection('## Full Command List');
|
||||
|
||||
const scores = await judge('command reference table', section);
|
||||
console.log('Command reference scores:', JSON.stringify(scores, null, 2));
|
||||
|
||||
// Completeness threshold is 3 (not 4) — the command reference table is
|
||||
// intentionally terse (quick-reference format). The judge consistently scores
|
||||
// completeness=3 because detailed argument docs live in per-command sections.
|
||||
evalCollector?.addTest({
|
||||
name: 'command reference table',
|
||||
suite: 'LLM-as-judge quality evals',
|
||||
tier: 'llm-judge',
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: scores.reasoning,
|
||||
});
|
||||
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
expect(scores.completeness).toBeGreaterThanOrEqual(3);
|
||||
expect(scores.actionability).toBeGreaterThanOrEqual(4);
|
||||
}, JUDGE_MS);
|
||||
|
||||
testIfSelected('snapshot flags reference', async () => {
|
||||
const t0 = Date.now();
|
||||
// Browse carve: snapshot flags live in browse/sections/command-list.md now,
|
||||
// ordered before '## Full Command List' (the '## CSS Inspector' end boundary
|
||||
// stayed in the skeleton).
|
||||
const section = sliceBrowseSection('## Snapshot Flags', '## Full Command List');
|
||||
|
||||
const scores = await judge('snapshot flags reference', section);
|
||||
console.log('Snapshot flags scores:', JSON.stringify(scores, null, 2));
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'snapshot flags reference',
|
||||
suite: 'LLM-as-judge quality evals',
|
||||
tier: 'llm-judge',
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: scores.reasoning,
|
||||
});
|
||||
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
expect(scores.completeness).toBeGreaterThanOrEqual(4);
|
||||
expect(scores.actionability).toBeGreaterThanOrEqual(4);
|
||||
}, JUDGE_MS);
|
||||
|
||||
testIfSelected('browse/SKILL.md reference', async () => {
|
||||
const t0 = Date.now();
|
||||
// Browse carve: flags + commands are the whole generated section file.
|
||||
// Browse carve: snapshot flags + the full command list are the whole
|
||||
// generated section file; one judge grades the union. Scores are also
|
||||
// pinned against test/fixtures/eval-baselines.json (UPDATE_BASELINES=1
|
||||
// rewrites the pin).
|
||||
const section = sliceBrowseSection('## Snapshot Flags');
|
||||
|
||||
const scores = await judge('browse skill reference (flags + commands)', section);
|
||||
console.log('Browse SKILL.md scores:', JSON.stringify(scores, null, 2));
|
||||
|
||||
const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json');
|
||||
const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8'));
|
||||
const regressions = (['clarity', 'completeness', 'actionability'] as const)
|
||||
.filter(dim => scores[dim] < baselines.browse_skill[dim])
|
||||
.map(dim => `browse_skill.${dim}: ${scores[dim]} < baseline ${baselines.browse_skill[dim]}`);
|
||||
if (process.env.UPDATE_BASELINES) {
|
||||
baselines.browse_skill = { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability };
|
||||
fs.writeFileSync(baselinesPath, JSON.stringify(baselines, null, 2) + '\n');
|
||||
console.log('Updated eval baselines');
|
||||
}
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'browse/SKILL.md reference',
|
||||
suite: 'LLM-as-judge quality evals',
|
||||
tier: 'llm-judge',
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4,
|
||||
passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4 && regressions.length === 0,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability },
|
||||
judge_reasoning: scores.reasoning,
|
||||
judge_reasoning: regressions.length ? `${scores.reasoning} | ${regressions.join('; ')}` : scores.reasoning,
|
||||
});
|
||||
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
expect(scores.completeness).toBeGreaterThanOrEqual(4);
|
||||
expect(scores.actionability).toBeGreaterThanOrEqual(4);
|
||||
expect(regressions).toEqual([]);
|
||||
}, JUDGE_MS);
|
||||
|
||||
testIfSelected('setup block', async () => {
|
||||
@@ -205,89 +163,6 @@ describeIfSelected('LLM-as-judge quality evals', [
|
||||
expect(scores.clarity).toBeGreaterThanOrEqual(3);
|
||||
}, JUDGE_MS);
|
||||
|
||||
testIfSelected('regression vs baseline', async () => {
|
||||
const t0 = Date.now();
|
||||
// Browse carve: the command reference lives in browse/sections/command-list.md.
|
||||
const genSection = sliceBrowseSection('## Full Command List');
|
||||
|
||||
const baseline = `## Command Reference
|
||||
|
||||
### Navigation
|
||||
| Command | Description |
|
||||
|---------|-------------|
|
||||
| \`goto <url>\` | Navigate to URL |
|
||||
| \`back\` / \`forward\` | History navigation |
|
||||
| \`reload\` | Reload page |
|
||||
| \`url\` | Print current URL |
|
||||
|
||||
### Interaction
|
||||
| Command | Description |
|
||||
|---------|-------------|
|
||||
| \`click <sel>\` | Click element |
|
||||
| \`fill <sel> <val>\` | Fill input |
|
||||
| \`select <sel> <val>\` | Select dropdown |
|
||||
| \`hover <sel>\` | Hover element |
|
||||
| \`type <text>\` | Type into focused element |
|
||||
| \`press <key>\` | Press key (Enter, Tab, Escape) |
|
||||
| \`scroll [sel]\` | Scroll element into view |
|
||||
| \`wait <sel>\` | Wait for element (max 10s) |
|
||||
| \`wait --networkidle\` | Wait for network to be idle |
|
||||
| \`wait --load\` | Wait for page load event |
|
||||
|
||||
### Inspection
|
||||
| Command | Description |
|
||||
|---------|-------------|
|
||||
| \`js <expr>\` | Run JavaScript |
|
||||
| \`css <sel> <prop>\` | Computed CSS |
|
||||
| \`attrs <sel>\` | Element attributes |
|
||||
| \`is <prop> <sel>\` | State check (visible/hidden/enabled/disabled/checked/editable/focused) |
|
||||
| \`console [--clear\\|--errors]\` | Console messages (--errors filters to error/warning) |`;
|
||||
|
||||
const client = new Anthropic();
|
||||
const response = await client.messages.create({
|
||||
model: 'claude-sonnet-4-6',
|
||||
max_tokens: 1024,
|
||||
messages: [{
|
||||
role: 'user',
|
||||
content: `You are comparing two versions of CLI documentation for an AI coding agent.
|
||||
|
||||
VERSION A (baseline — hand-maintained):
|
||||
${baseline}
|
||||
|
||||
VERSION B (auto-generated from source):
|
||||
${genSection}
|
||||
|
||||
Which version is better for an AI agent trying to use these commands? Consider:
|
||||
- Completeness (more commands documented? all args shown?)
|
||||
- Clarity (descriptions helpful?)
|
||||
- Coverage (missing commands in either version?)
|
||||
|
||||
Respond with ONLY valid JSON:
|
||||
{"winner": "A" or "B" or "tie", "reasoning": "brief explanation", "a_score": N, "b_score": N}
|
||||
|
||||
Scores are 1-5 overall quality.`,
|
||||
}],
|
||||
});
|
||||
|
||||
const text = response.content[0].type === 'text' ? response.content[0].text : '';
|
||||
const jsonMatch = text.match(/\{[\s\S]*\}/);
|
||||
if (!jsonMatch) throw new Error(`Judge returned non-JSON: ${text.slice(0, 200)}`);
|
||||
const result = JSON.parse(jsonMatch[0]);
|
||||
console.log('Regression comparison:', JSON.stringify(result, null, 2));
|
||||
|
||||
evalCollector?.addTest({
|
||||
name: 'regression vs baseline',
|
||||
suite: 'LLM-as-judge quality evals',
|
||||
tier: 'llm-judge',
|
||||
passed: result.b_score >= result.a_score,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
judge_scores: { a_score: result.a_score, b_score: result.b_score },
|
||||
judge_reasoning: result.reasoning,
|
||||
});
|
||||
|
||||
expect(result.b_score).toBeGreaterThanOrEqual(result.a_score);
|
||||
}, JUDGE_MS);
|
||||
});
|
||||
|
||||
// --- Part 7: QA skill quality evals (C6) ---
|
||||
@@ -523,59 +398,6 @@ score (1-5): 5 = perfectly consistent, 1 = contradictory`);
|
||||
}, JUDGE_MS);
|
||||
});
|
||||
|
||||
// --- Part 7: Baseline score pinning (C9) ---
|
||||
|
||||
describeIfSelected('Baseline score pinning', ['baseline score pinning'], () => {
|
||||
const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json');
|
||||
|
||||
testIfSelected('baseline score pinning', async () => {
|
||||
const t0 = Date.now();
|
||||
if (!fs.existsSync(baselinesPath)) {
|
||||
console.log('No baseline file found — skipping pinning check');
|
||||
return;
|
||||
}
|
||||
|
||||
const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8'));
|
||||
const regressions: string[] = [];
|
||||
|
||||
// Browse carve: the command reference lives in browse/sections/command-list.md.
|
||||
const cmdSection = sliceBrowseSection('## Full Command List');
|
||||
const cmdScores = await judge('command reference table', cmdSection);
|
||||
|
||||
for (const dim of ['clarity', 'completeness', 'actionability'] as const) {
|
||||
if (cmdScores[dim] < baselines.command_reference[dim]) {
|
||||
regressions.push(`command_reference.${dim}: ${cmdScores[dim]} < baseline ${baselines.command_reference[dim]}`);
|
||||
}
|
||||
}
|
||||
|
||||
if (process.env.UPDATE_BASELINES) {
|
||||
baselines.command_reference = {
|
||||
clarity: cmdScores.clarity,
|
||||
completeness: cmdScores.completeness,
|
||||
actionability: cmdScores.actionability,
|
||||
};
|
||||
fs.writeFileSync(baselinesPath, JSON.stringify(baselines, null, 2) + '\n');
|
||||
console.log('Updated eval baselines');
|
||||
}
|
||||
|
||||
const passed = regressions.length === 0;
|
||||
evalCollector?.addTest({
|
||||
name: 'baseline score pinning',
|
||||
suite: 'Baseline score pinning',
|
||||
tier: 'llm-judge',
|
||||
passed,
|
||||
duration_ms: Date.now() - t0,
|
||||
cost_usd: 0.02,
|
||||
judge_scores: { clarity: cmdScores.clarity, completeness: cmdScores.completeness, actionability: cmdScores.actionability },
|
||||
judge_reasoning: passed ? 'All scores at or above baseline' : regressions.join('; '),
|
||||
});
|
||||
|
||||
if (!passed) {
|
||||
throw new Error(`Score regressions detected:\n${regressions.join('\n')}`);
|
||||
}
|
||||
}, JUDGE_MS);
|
||||
});
|
||||
|
||||
// --- Workflow SKILL.md quality evals (10 new tests for 100% coverage) ---
|
||||
|
||||
/**
|
||||
|
||||
@@ -590,6 +590,52 @@ export default app;
|
||||
}
|
||||
}, CAPTURE_MS);
|
||||
|
||||
testIfSelected('journey-negatives', async () => {
|
||||
// Casual or off-topic prompts that share routing keywords ("wtf",
|
||||
// "algorithm", "send it to the team") must not invoke a skill. Folded
|
||||
// from the retired Opus 4.7 routing eval; same bound: at most one of the
|
||||
// three may route.
|
||||
const cases = [
|
||||
{ name: 'neg-syntax-q', prompt: 'wtf does this Python list comprehension syntax even mean, [x for x in y if z]?' },
|
||||
{ name: 'neg-algo-q', prompt: 'does this bubble sort algorithm actually work in O(n log n)?' },
|
||||
{ name: 'neg-slack-send', prompt: 'can you help me write the slack message? I want to send it to the team.' },
|
||||
];
|
||||
const tmpDir = createRoutingWorkDir('negatives');
|
||||
try {
|
||||
const results = await Promise.all(cases.map(async c => {
|
||||
const result = await runSkillTest({
|
||||
prompt: c.prompt,
|
||||
workingDirectory: tmpDir,
|
||||
maxTurns: 2,
|
||||
allowedTools: ['Skill', 'Read'],
|
||||
timeout: JUDGE_MS,
|
||||
testName: `journey-negatives-${c.name}`,
|
||||
runId,
|
||||
});
|
||||
const skillCalls = result.toolCalls.filter(tc => tc.tool === 'Skill');
|
||||
const actualSkill = skillCalls.length > 0 ? skillCalls[0]?.input?.skill : undefined;
|
||||
logCost(`journey: journey-negatives ${c.name}`, result);
|
||||
evalCollector?.addTest({
|
||||
name: `journey-negatives-${c.name}`,
|
||||
suite: 'Skill Routing E2E',
|
||||
tier: 'e2e',
|
||||
passed: actualSkill === undefined,
|
||||
duration_ms: result.duration,
|
||||
cost_usd: result.costEstimate.estimatedCost,
|
||||
transcript: result.transcript,
|
||||
output: `routed=${actualSkill ?? '(none)'}`,
|
||||
turns_used: result.costEstimate.turnsUsed,
|
||||
exit_reason: result.exitReason,
|
||||
});
|
||||
return { name: c.name, actualSkill };
|
||||
}));
|
||||
const routed = results.filter(r => r.actualSkill !== undefined);
|
||||
expect(routed.length, `negatives routed: ${routed.map(r => `${r.name}→${r.actualSkill}`).join(', ')}`).toBeLessThanOrEqual(1);
|
||||
} finally {
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||||
}
|
||||
}, CAPTURE_MS);
|
||||
|
||||
testIfSelected('journey-visual-qa', async () => {
|
||||
const tmpDir = createRoutingWorkDir('visual-qa');
|
||||
try {
|
||||
|
||||
@@ -181,7 +181,6 @@ describe('test-free-shards: enumeration', () => {
|
||||
expect(isFreeTestFile('test/skill-llm-eval.test.ts')).toBe(false);
|
||||
expect(isFreeTestFile('test/codex-e2e.test.ts')).toBe(false);
|
||||
expect(isFreeTestFile('test/codex-e2e-sol-scope.test.ts')).toBe(false);
|
||||
expect(isFreeTestFile('test/gemini-e2e.test.ts')).toBe(false);
|
||||
});
|
||||
|
||||
test('collectFreeTestFiles returns sorted, deduped, only-free list', () => {
|
||||
|
||||
+6
-14
@@ -83,8 +83,7 @@ describe('selectTests', () => {
|
||||
// These cases use an outside-only/Code Quality excerpt or descriptive metadata.
|
||||
.filter(id => !['outside-plan-disabled-no-fallback', 'plan-ceo-review-prosons-cadence',
|
||||
'plan-review-prosons-format', 'shared-libs-plan-callers'].includes(id));
|
||||
const existing = ['learnings-show', 'codex-plan-ceo-format-mode', 'codex-plan-ceo-format-approach',
|
||||
'codex-plan-eng-format-coverage', 'codex-plan-eng-format-kind'];
|
||||
const existing = ['learnings-show'];
|
||||
const result = selectTests(['scripts/resolvers/learnings.ts'], E2E_TOUCHFILES);
|
||||
expect(result.reason).toBe('diff');
|
||||
expect(result.selected.sort()).toEqual([...new Set([...consumers, ...existing])].sort());
|
||||
@@ -161,12 +160,8 @@ describe('selectTests', () => {
|
||||
expect(fs.readFileSync(path.join(ROOT, `${output}.tmpl`), 'utf8')).toContain(`{{${token}}}`);
|
||||
}
|
||||
const generated = selectTests(consumers.map(([output]) => output), E2E_TOUCHFILES);
|
||||
// These two CEO-format cases already depend on every resolver through
|
||||
// scripts/resolvers/**; keep that existing selection alongside consumers.
|
||||
// The bounded Code Quality fixture stops before Test review.
|
||||
const expected = [...new Set([...generated.selected.filter(id => id !== 'shared-libs-plan-callers' && !SHIP_GUARD_ONLY.includes(id)),
|
||||
'codex-plan-ceo-format-mode', 'codex-plan-ceo-format-approach',
|
||||
])].sort();
|
||||
const expected = [...new Set(generated.selected.filter(id => id !== 'shared-libs-plan-callers' && !SHIP_GUARD_ONLY.includes(id)))].sort();
|
||||
const actual = selectTests(['scripts/resolvers/testing.ts'], E2E_TOUCHFILES);
|
||||
expect(actual.reason).toBe('diff');
|
||||
expect(actual.selected.sort()).toEqual(expected);
|
||||
@@ -373,11 +368,9 @@ describe('selectTests', () => {
|
||||
expect(result.selected).toContain('plan-ceo-split-overflow');
|
||||
// v2 plan Phase B carve: the section-loading E2E depends on plan-ceo-review/**.
|
||||
expect(result.selected).toContain('plan-ceo-section-loading');
|
||||
expect(result.selected).toContain('codex-plan-ceo-format-mode');
|
||||
expect(result.selected).toContain('codex-plan-ceo-format-approach');
|
||||
expect(result.selected).toContain('outside-plan-disabled-no-fallback');
|
||||
expect(result.selected.length).toBe(25);
|
||||
expect(result.skipped.length).toBe(Object.keys(E2E_TOUCHFILES).length - 25);
|
||||
expect(result.selected.length).toBe(23);
|
||||
expect(result.skipped.length).toBe(Object.keys(E2E_TOUCHFILES).length - 23);
|
||||
});
|
||||
|
||||
test('global touchfile triggers ALL tests', () => {
|
||||
@@ -398,7 +391,6 @@ describe('selectTests', () => {
|
||||
expect(result.reason).toBe('diff');
|
||||
expect(result.selected).toContain('autoplan-chain-pty');
|
||||
expect(result.selected).toContain('plan-ceo-mode-routing');
|
||||
expect(result.selected).not.toContain('codex-plan-ceo-format-mode');
|
||||
expect(result.selected).not.toContain('retro');
|
||||
});
|
||||
|
||||
@@ -604,7 +596,7 @@ describe('TOUCHFILES completeness', () => {
|
||||
);
|
||||
|
||||
const unique = registeredJudgeTestNames(llmContent);
|
||||
expect(unique).toHaveLength(27);
|
||||
expect(unique).toHaveLength(23);
|
||||
|
||||
const missing = unique.filter(name => !(name in LLM_JUDGE_TOUCHFILES));
|
||||
if (missing.length > 0) {
|
||||
@@ -623,7 +615,7 @@ describe('TOUCHFILES completeness', () => {
|
||||
testIfSelected('unmapped judge case', async () => {}, 120_000);
|
||||
`;
|
||||
const names = registeredJudgeTestNames(withUnmappedCase);
|
||||
expect(names).toHaveLength(28);
|
||||
expect(names).toHaveLength(24);
|
||||
expect(names.filter(name => !(name in LLM_JUDGE_TOUCHFILES))).toEqual(['unmapped judge case']);
|
||||
});
|
||||
|
||||
|
||||
@@ -163,7 +163,7 @@ test('workflow registration preserves model work and reserves only terminal-reco
|
||||
expect(source).toContain('const WORKFLOW_JUDGE_RECORD_MS = 5_000;');
|
||||
expect(source).toContain('const WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000;');
|
||||
expect(source.match(/\}, WORKFLOW_JUDGE_TEST_MS\);/g)).toHaveLength(16);
|
||||
expect(source.match(/\}, JUDGE_MS\);/g)).toHaveLength(11);
|
||||
expect(source.match(/\}, JUDGE_MS\);/g)).toHaveLength(7);
|
||||
});
|
||||
|
||||
function actualCallback(f: ReturnType<typeof fixture>, overrides: {
|
||||
|
||||
Reference in new issue
Block a user