diff --git a/.gitignore b/.gitignore index 015af1ed3..720242426 100644 --- a/.gitignore +++ b/.gitignore @@ -59,6 +59,6 @@ docs/throughput-*.json .sources/ # SPM build output from the gen-accessors tool (built in place by -# skill-e2e-ios-swift-build; regenerates on every run — never commit) +# test/ios-qa-swift-build.test.ts; regenerates on every run — never commit) ios-qa/scripts/gen-accessors-tool/.build/ ios-qa/scripts/gen-accessors-tool/Package.resolved diff --git a/TODOS.md b/TODOS.md index e44efb428..707e32d7a 100644 --- a/TODOS.md +++ b/TODOS.md @@ -502,32 +502,6 @@ touchfiles and re-offer pending ones on the next interactive run. false) permanently misses the artifacts-rename migration unless they paste the manual command. **Effort:** M. **Priority:** P2. -### P2: periodic tier — TWO documented-red tests need structural repair (was three) - -**2026-08-29 update (test-infra overhaul):** (1) the sidebar E2E trio is -ALREADY DELETED — no file in the tree POSTs to /sidebar-command or -/sidebar-chat; only tombstone tests remain (browse/test/sidebar-tabs.test.ts -asserts the endpoints STAY deleted), so part (1) closes as already-done. -(2) skill-e2e-ship-idempotency and (3) skill-e2e-brain-privacy-gate are now -EXCLUDED from the weekly lane with tracking -(test/helpers/periodic-exclude-data.ts) — removing their entries re-activates -them; the structural investigations below are the re-entry condition. - -**What:** (1) The sidebar E2E trio (navigate, url-accuracy, css-interaction) -POSTs to /sidebar-command and /sidebar-chat — endpoints removed on every tree -when the PTY terminal replaced the chat queue (server.ts tombstone ~2671); -rewrite them against the PTY surface or delete them. (2) -skill-e2e-ship-idempotency: the PTY child sits at the Claude Code welcome -screen in plan mode for the full budget — the typed /ship never lands -(readiness/typing race vs CLI v2.1.233's welcome screen); never green since -it was born in v1.63. (3) skill-e2e-brain-privacy-gate: never green anywhere; -the artifacts-sync stop-gate preconditions don't survive the hermetic env -even with per-test HOME/GSTACK_HOME injection — needs a transcript-level -debug of what the child's preamble actually echoes. - -**Why:** every red periodic run costs triage time; two of these have burned -three triage passes across two releases. **Effort:** M. **Priority:** P2. - ### P1: #1882 — portable skill-install prefix (non-`gstack` install dirs break silently) **What:** Every generated SKILL.md hardcodes the literal `~/.claude/skills/gstack/...` @@ -851,6 +825,36 @@ audit trail lives in Aside. ## Test infrastructure +### P3: CI-unrunnable paid evals + +**What:** Seven paid files cannot execute in the CI image (no `codex` CLI, no +macOS/Aside, no physical iPhone), so the weekly periodic lane scheduled them as +green shards that verified nothing. They are now in `PERIODIC_CI_EXCLUDE` +(`test/helpers/periodic-exclude-data.ts`): `codex-e2e`, `codex-e2e-sol-scope`, +`codex-e2e-shared-libs`, `codex-e2e-recommendation-substance`, +`skill-e2e-outside-voice`, `skill-e2e-aside`, `skill-e2e-ios-device`. They still +run locally on a machine that has the CLI or device. + +**Re-entry:** the CLI or device is available in the CI image. First target: +`codex-e2e-sol-scope` as the Codex host smoke once the Codex CLI is installed +(see "Install the Codex CLI in the CI image"). Remove each file's exclude entry +when its prerequisite exists. + +**Review by:** 2026-12-28. **Effort:** S per file. **Priority:** P3. + +### P3: Install the Codex CLI in the CI image + +**What:** Add `@openai/codex` to `.github/docker/Dockerfile.ci` and provide a +Codex `auth.json` as a CI secret so the four `codex-e2e*` files and +`skill-e2e-outside-voice` can leave `PERIODIC_CI_EXCLUDE`. + +**Cost estimate:** image build +1 npm global install (~30 s per image build); +weekly model spend on the order of the repo's periodic rule of thumb, ~$1 per +file per run, so ~$5/week for the five files, billed to the Codex account +behind the secret. **Risk:** a long-lived credential in CI. + +**Effort:** S. **Priority:** P3. + ### P1: skillify gate test red — HOME-override sessions never discover project skills (pre-existing) **What:** `test/skill-e2e-skillify.test.ts` `skillify-provenance-refusal` fails @@ -980,9 +984,8 @@ coverage fill. Remaining, in rough priority order: - **P3 — eval-list should exclude _partial runs** (pinned as current behavior in test/eval-cli-family.test.ts with an improvement note). Effort S. -- **P3 — codex-e2e-plan-format's testIfSelected names have no map keys** - (run-all only today) + 15 E2E / 2 judge PHANTOM touchfiles keys select - tests that exist nowhere — add keys or delete, one sweep. Effort S. +- **P3 — 15 E2E / 2 judge PHANTOM touchfiles keys** select tests that exist + nowhere — add keys or delete, one sweep. Effort S. - **P3 — first-execution rot from the sliced lane's first live runs: 2 of 3 FIXED** (PR #2721): (a) ✅ skillify family — root cause was HOME==cwd making claude treat /.claude/skills as the PERSONAL dir (project @@ -3154,7 +3157,7 @@ files have no `evals.yml` matrix row, so CI never runs them (`KNOWN_MATRIX_GAPS` in the test enumerates them — notably the plan-mode and finding-floor smokes and the AUQ format-compliance gate). (2) Four matrix rows point at whole-file tier-gated files but set no row `tier:` property, so with -`EVALS_TIER` unexported those suites self-skip: `codex-e2e`/`gemini-e2e` run +`EVALS_TIER` unexported those suites self-skip: `codex-e2e` runs ZERO tests and report green on every PR (vestigial rows; the periodic cron lane owns them — consider deleting the rows), and `e2e-pty-plan-smoke` spends ~7 min on setup then skips every describe (hollow-green since the files diff --git a/package.json b/package.json index 60f2e55b3..2564f33f8 100644 --- a/package.json +++ b/package.json @@ -27,18 +27,16 @@ "test:free": "bun run scripts/test-free-shards.ts", "test:windows": "bun run scripts/test-free-shards.ts --windows-only", "test:ubicloud": "bash scripts/ubicloud/test-free.sh", - "test:evals": "EVALS=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/gemini-e2e.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", - "test:evals:all": "EVALS=1 EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/gemini-e2e.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", - "test:e2e": "EVALS=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/gemini-e2e.test.ts test/carve-section-loading*.test.ts", - "test:e2e:all": "EVALS=1 EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/gemini-e2e.test.ts test/carve-section-loading*.test.ts", - "test:gate": "EVALS=1 EVALS_TIER=gate bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/gemini-e2e.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", - "test:periodic": "EVALS=1 EVALS_TIER=periodic EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/gemini-e2e.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:evals": "EVALS=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:evals:all": "EVALS=1 EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:e2e": "EVALS=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts", + "test:e2e:all": "EVALS=1 EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/carve-section-loading*.test.ts", + "test:gate": "EVALS=1 EVALS_TIER=gate bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", + "test:periodic": "EVALS=1 EVALS_TIER=periodic EVALS_ALL=1 bun test --retry 1 --concurrent --max-concurrency ${EVALS_CONCURRENCY:-15} test/skill-llm-eval*.test.ts test/skill-e2e-*.test.ts test/skill-routing-e2e.test.ts test/codex-e2e*.test.ts test/llm-judge-recommendation.test.ts test/carve-section-loading*.test.ts", "test:gate:sharded": "bun run scripts/test-paid-shards.ts --tier gate", "test:periodic:sharded": "EVALS_ALL=1 bun run scripts/test-paid-shards.ts --tier periodic", "test:codex": "EVALS=1 bun test test/codex-e2e.test.ts test/codex-e2e-sol-scope.test.ts", "test:codex:all": "EVALS=1 EVALS_ALL=1 bun test test/codex-e2e.test.ts test/codex-e2e-sol-scope.test.ts", - "test:gemini": "EVALS=1 bun test test/gemini-e2e.test.ts", - "test:gemini:all": "EVALS=1 EVALS_ALL=1 bun test test/gemini-e2e.test.ts", "skill:check": "bun run scripts/skill-check.ts", "dev:skill": "bun run scripts/dev-skill.ts", "start": "bun run browse/src/server.ts", diff --git a/scripts/paid-test-durations.json b/scripts/paid-test-durations.json index d6d5e11d5..47cbdedd6 100644 --- a/scripts/paid-test-durations.json +++ b/scripts/paid-test-durations.json @@ -15,14 +15,12 @@ "test/skill-e2e-investigate-owned-termination.test.ts": 49000, "test/skill-e2e-learnings.test.ts": 32000, "test/skill-e2e-office-hours-auto-mode.test.ts": 61000, - "test/skill-e2e-opus-47.test.ts": 20000, "test/skill-e2e-plan-ceo-finding-floor.test.ts": 233000, "test/skill-e2e-plan-ceo-plan-mode.test.ts": 35000, "test/skill-e2e-plan-design-with-ui.test.ts": 435000, "test/skill-e2e-plan-devex-finding-floor.test.ts": 187000, "test/skill-e2e-plan-devex-plan-mode.test.ts": 103000, "test/skill-e2e-plan-mode-no-op.test.ts": 206000, - "test/skill-e2e-plan-tune-cathedral.test.ts": 1000, "test/skill-e2e-plan-tune.test.ts": 58000, "test/skill-e2e-plan.test.ts": 312000, "test/skill-e2e-qa-workflow.test.ts": 437000, diff --git a/scripts/test-paid-shards.ts b/scripts/test-paid-shards.ts index 69ced5bc2..8feaf4d47 100644 --- a/scripts/test-paid-shards.ts +++ b/scripts/test-paid-shards.ts @@ -393,7 +393,7 @@ export interface DiffSkipOptions { * than literal. * * FAIL-OPEN by construction: run-all selection, non-skill-e2e paid files - * (llm-judge / codex-e2e / gemini-e2e / routing, keyed off other maps), + * (llm-judge / codex-e2e / routing, keyed off other maps), * unreadable sources, and files with zero mapped names all KEEP their shard — * the child's self-skip stays authoritative. A parent bug may only run * extra work, never drop it. diff --git a/test/auto-decide-saved-ai.test.ts b/test/auto-decide-saved-ai.test.ts index 5906ed268..673b1f257 100644 --- a/test/auto-decide-saved-ai.test.ts +++ b/test/auto-decide-saved-ai.test.ts @@ -50,7 +50,7 @@ test('failed loads, foreign sessions, actual questions and later withdrawals ret import { E2E_TOUCHFILES } from './helpers/touchfiles-data'; test('new evidence inputs retain every native observation caller',()=>{ - const owners=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','auto-decide-preserved','conductor-prose']; + const owners=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','auto-decide-preserved']; for(const file of ['test/auto-decide-saved-ai.test.ts','test/fixtures/auto-decide-saved-ai.json','test/fixtures/auto-decide-retry-ai.json']) expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(file)).map(([name])=>name)).toEqual(owners); }); diff --git a/test/autoplan-public-narration.test.ts b/test/autoplan-public-narration.test.ts index 7dfeaacbd..92bf13593 100644 --- a/test/autoplan-public-narration.test.ts +++ b/test/autoplan-public-narration.test.ts @@ -127,7 +127,7 @@ test('phase ordering and duplicate collapse use native time rather than polling test('public narration changes select every existing shared native-reader consumer',()=>{ const expected=[ - 'auto-decide-preserved','autoplan-chain-pty','conductor-prose', + 'auto-decide-preserved','autoplan-chain-pty', 'plan-ceo-finding-count','plan-ceo-mode-routing','plan-ceo-split-overflow', 'plan-design-finding-count','plan-design-review-plan-mode','plan-design-with-ui-scope', 'plan-devex-finding-count','plan-eng-finding-count','plan-eng-multi-finding-batching', diff --git a/test/codex-e2e-plan-format.test.ts b/test/codex-e2e-plan-format.test.ts deleted file mode 100644 index 6c0169cfc..000000000 --- a/test/codex-e2e-plan-format.test.ts +++ /dev/null @@ -1,289 +0,0 @@ -/** - * AskUserQuestion format regression test for /plan-ceo-review and /plan-eng-review - * running under Codex CLI (GPT-5.4). - * - * Context: GPT-class models under the "No preamble / Prefer doing over listing" - * gpt.md overlay tend to skip the Simplify (ELI10) paragraph and the RECOMMENDATION - * line on AskUserQuestion calls. The user has to manually re-prompt "ELI10 and don't - * forget to recommend" almost every time. This test pins that behavior so future - * regressions surface automatically. - * - * Mirrors test/skill-e2e-plan-format.test.ts (the Claude version) but uses - * test/helpers/codex-session-runner.ts to drive `codex exec` instead of `claude -p`. - * - * Four cases: - * 1. plan-ceo-review mode selection (kind-differentiated) - * 2. plan-ceo-review approach menu (coverage-differentiated) - * 3. plan-eng-review per-issue coverage decision - * 4. plan-eng-review per-issue architectural choice (kind-differentiated) - * - * Assertions on captured AskUserQuestion text: - * - RECOMMENDATION: Choose present (all cases) - * - Completeness: N/10 present on coverage cases, absent on kind cases - * - "options differ in kind" note present on kind cases - * - ELI10-style plain-English explanation present (length floor + no raw jargon) - * - * Periodic tier (Codex non-determinism). Cost: ~$2-3 per full run. - */ -import { describe, test, beforeAll, afterAll } from 'bun:test'; -import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets'; -import { runCodexSkill } from './helpers/codex-session-runner'; -import { CODEX_EVAL_FINALIZE_MS, createCodexEvalCollector, runRecordedCodexEval, createCodexPlanFormatCapture } from './helpers/codex-eval'; -import { selectTests, detectBaseBranch, getChangedFiles, E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from './helpers/touchfiles'; -import * as fs from 'fs'; -import * as path from 'path'; -import * as os from 'os'; -import { spawnSync } from 'child_process'; - -const ROOT = path.resolve(import.meta.dir, '..'); - -// --- Prerequisites --- - -const CODEX_AVAILABLE = (() => { - try { - const result = Bun.spawnSync(['which', 'codex'], { timeout: 30_000 }); - return result.exitCode === 0; - } catch { return false; } -})(); -const evalsEnabled = !!process.env.EVALS; -// External-service test — periodic tier only (CLAUDE.md tiering rule 3), -// matching codex-e2e.test.ts / codex-e2e-sol-scope.test.ts. Without this -// guard the sharded runner's "no whole-file tier guard" default would run -// Codex spawns in the GATE tier on every PR. -const tierOk = process.env.EVALS_TIER === 'periodic'; -const SKIP = !CODEX_AVAILABLE || !evalsEnabled || !tierOk; -const describeCodex = SKIP ? describe.skip : describe; - -// --- Touchfiles --- - -// Keep selection dependencies in the canonical map, including the test helpers. -const CODEX_FORMAT_TOUCHFILES: Record = Object.fromEntries( - ['codex-plan-ceo-format-mode', 'codex-plan-ceo-format-approach', - 'codex-plan-eng-format-coverage', 'codex-plan-eng-format-kind'].map((key) => { - if (!E2E_TOUCHFILES[key]) throw new Error(`canonical E2E_TOUCHFILES lost key '${key}'`); - return [key, E2E_TOUCHFILES[key]]; - }), -); - -let selectedTests: string[] | null = null; -if (evalsEnabled && !process.env.EVALS_ALL) { - const baseBranch = process.env.EVALS_BASE || detectBaseBranch(ROOT) || 'main'; - const changedFiles = getChangedFiles(baseBranch, ROOT); - if (changedFiles.length > 0) { - const selection = selectTests(changedFiles, CODEX_FORMAT_TOUCHFILES, GLOBAL_TOUCHFILES); - selectedTests = selection.selected; - } -} - -function testIfSelected(name: string, fn: () => Promise, timeout: number) { - if (selectedTests !== null && !selectedTests.includes(name)) { - test.skip(name, fn, timeout + CODEX_EVAL_FINALIZE_MS); - } else { - test(name, fn, timeout + CODEX_EVAL_FINALIZE_MS); - } -} - -// --- Eval collector --- - -const evalCollector = SKIP ? null : createCodexEvalCollector('codex-e2e-plan-format'); - -afterAll(async () => { - if (evalCollector) { - await evalCollector.finalize(); - } -}); - -// --- Fixtures --- - -const SAMPLE_PLAN = `# Plan: Add User Dashboard - -## Context -We're building a new user dashboard that shows recent activity, notifications, and quick actions. - -## Changes -1. New React component \`UserDashboard\` in \`src/components/\` -2. REST API endpoint \`GET /api/dashboard\` returning user stats -3. PostgreSQL query for activity aggregation -4. Redis cache layer for dashboard data (5min TTL) - -## Architecture -- Frontend: React + TailwindCSS -- Backend: Express.js REST API -- Database: PostgreSQL with existing user/activity tables -- Cache: Redis for dashboard aggregates -`; - -function setupCodexSkillDir(tmpPrefix: string, skillName: 'plan-ceo-review' | 'plan-eng-review'): { skillDir: string; planDir: string; outFile: string } { - const planDir = fs.mkdtempSync(path.join(os.tmpdir(), tmpPrefix)); - const run = (cmd: string, args: string[]) => - spawnSync(cmd, args, { cwd: planDir, stdio: 'pipe', timeout: 5000 }); - - run('git', ['init', '-b', 'main']); - run('git', ['config', 'user.email', 'test@test.com']); - run('git', ['config', 'user.name', 'Test']); - - fs.writeFileSync(path.join(planDir, 'plan.md'), SAMPLE_PLAN); - run('git', ['add', '.']); - run('git', ['commit', '-m', 'add plan']); - - // Codex skill lives in .agents/skills/gstack-{name}/ per the gstack host convention. - const codexSkillSource = path.join(ROOT, '.agents', 'skills', `gstack-${skillName}`); - const skillDir = path.join(planDir, '.agents', 'skills', `gstack-${skillName}`); - fs.mkdirSync(skillDir, { recursive: true }); - fs.cpSync(codexSkillSource, skillDir, { recursive: true }); - - const outFile = path.join(planDir, 'ask-capture.md'); - return { skillDir, planDir, outFile }; -} - -// Capture instruction — same shape as the Claude version. Codex may ignore tool calls, -// so we tell it to write prose to the file directly. -function captureInstruction(outFile: string): string { - return `Write the verbatim text of every AskUserQuestion you would have presented to the user to the file ${outFile} (one question per session, full text including the re-ground, ELI10 paragraph, RECOMMENDATION line, and options). Do NOT ask the user interactively. Do NOT paraphrase. This is a format-capture test, not an interactive session.`; -} - -// --- Tests --- - -describeCodex('Codex Plan Format — CEO Mode Selection', () => { - let skillDir: string, planDir: string, outFile: string; - - beforeAll(() => { - ({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-ceo-mode-', 'plan-ceo-review')); - }); - - afterAll(() => { - try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {} - }); - - testIfSelected('codex-plan-ceo-format-mode', async () => { - const capture = createCodexPlanFormatCapture(outFile, 'kind'); - const result = await runRecordedCodexEval({ - name: 'codex-plan-ceo-format-mode', - suite: 'codex-e2e-plan-format', - budgetMs: CAPTURE_LONG_MS, - run: (signal) => { - capture.reset(); - return runCodexSkill({ - skillDir, - prompt: `Read the plan-ceo-review skill. Read plan.md (the plan to review). Proceed to Mode Selection where the skill presents 4 mode options (SCOPE EXPANSION, SELECTIVE EXPANSION, HOLD SCOPE, SCOPE REDUCTION) via AskUserQuestion. These options differ in kind (review posture), not coverage. ${captureInstruction(outFile)}`, - timeoutMs: CAPTURE_MS, - cwd: planDir, - skillName: 'gstack-plan-ceo-review', - sandbox: 'workspace-write', - signal, - }); - }, - validate: capture.validate, - record: (entry) => evalCollector?.addTest(capture.attach(entry)), - }); - console.log(`codex-plan-ceo-format-mode: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`); - }, CAPTURE_LONG_MS); -}); - -describeCodex('Codex Plan Format — CEO Approach Menu', () => { - let skillDir: string, planDir: string, outFile: string; - - beforeAll(() => { - ({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-ceo-approach-', 'plan-ceo-review')); - }); - - afterAll(() => { - try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {} - }); - - testIfSelected('codex-plan-ceo-format-approach', async () => { - const capture = createCodexPlanFormatCapture(outFile, 'coverage'); - const result = await runRecordedCodexEval({ - name: 'codex-plan-ceo-format-approach', - suite: 'codex-e2e-plan-format', - budgetMs: CAPTURE_LONG_MS, - run: (signal) => { - capture.reset(); - return runCodexSkill({ - skillDir, - prompt: `Read the plan-ceo-review skill. Read plan.md. Proceed to Alternatives (the implementation approach menu) where the skill generates 2-3 approaches (minimal viable vs ideal architecture) and presents them via AskUserQuestion. These options differ in coverage so Completeness: N/10 applies. ${captureInstruction(outFile)}`, - timeoutMs: CAPTURE_MS, - cwd: planDir, - skillName: 'gstack-plan-ceo-review', - sandbox: 'workspace-write', - signal, - }); - }, - validate: capture.validate, - record: (entry) => evalCollector?.addTest(capture.attach(entry)), - }); - console.log(`codex-plan-ceo-format-approach: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`); - }, CAPTURE_LONG_MS); -}); - -describeCodex('Codex Plan Format — Eng Coverage Issue', () => { - let skillDir: string, planDir: string, outFile: string; - - beforeAll(() => { - ({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-eng-cov-', 'plan-eng-review')); - }); - - afterAll(() => { - try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {} - }); - - testIfSelected('codex-plan-eng-format-coverage', async () => { - const capture = createCodexPlanFormatCapture(outFile, 'coverage'); - const result = await runRecordedCodexEval({ - name: 'codex-plan-eng-format-coverage', - suite: 'codex-e2e-plan-format', - budgetMs: CAPTURE_LONG_MS, - run: (signal) => { - capture.reset(); - return runCodexSkill({ - skillDir, - prompt: `Read the plan-eng-review skill. Read plan.md. In your Section 3 Test Review, generate ONE AskUserQuestion about test coverage depth where options are clearly coverage-differentiated: A) full coverage incl. edge + error paths (Completeness 10/10), B) happy path only (7/10), C) smoke test (3/10). ${captureInstruction(outFile)}`, - timeoutMs: CAPTURE_MS, - cwd: planDir, - skillName: 'gstack-plan-eng-review', - sandbox: 'workspace-write', - signal, - }); - }, - validate: capture.validate, - record: (entry) => evalCollector?.addTest(capture.attach(entry)), - }); - console.log(`codex-plan-eng-format-coverage: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`); - }, CAPTURE_LONG_MS); -}); - -describeCodex('Codex Plan Format — Eng Kind Issue', () => { - let skillDir: string, planDir: string, outFile: string; - - beforeAll(() => { - ({ skillDir, planDir, outFile } = setupCodexSkillDir('codex-e2e-plan-format-eng-kind-', 'plan-eng-review')); - }); - - afterAll(() => { - try { fs.rmSync(planDir, { recursive: true, force: true }); } catch {} - }); - - testIfSelected('codex-plan-eng-format-kind', async () => { - const capture = createCodexPlanFormatCapture(outFile, 'kind'); - const result = await runRecordedCodexEval({ - name: 'codex-plan-eng-format-kind', - suite: 'codex-e2e-plan-format', - budgetMs: CAPTURE_LONG_MS, - run: (signal) => { - capture.reset(); - return runCodexSkill({ - skillDir, - prompt: `Read the plan-eng-review skill. Read plan.md. In your Section 1 Architecture review, generate ONE AskUserQuestion about an architectural choice where the options differ in kind (e.g. Redis vs Postgres materialized view vs in-process cache — different kinds of systems with different tradeoffs, NOT more-or-less-complete versions of the same thing). ${captureInstruction(outFile)}`, - timeoutMs: CAPTURE_MS, - cwd: planDir, - skillName: 'gstack-plan-eng-review', - sandbox: 'workspace-write', - signal, - }); - }, - validate: capture.validate, - record: (entry) => evalCollector?.addTest(capture.attach(entry)), - }); - console.log(`codex-plan-eng-format-kind: ${result.tokens}t, ${Math.round(result.durationMs/1000)}s, exit=${result.exitCode}`); - }, CAPTURE_LONG_MS); -}); diff --git a/test/codex-eval-selection.test.ts b/test/codex-eval-selection.test.ts index b0bcfa379..d19e16ce7 100644 --- a/test/codex-eval-selection.test.ts +++ b/test/codex-eval-selection.test.ts @@ -11,23 +11,11 @@ describe('Codex eval selection', () => { expect(selected.sort()).toEqual([ 'codex-discover-skill', 'codex-review-findings', - 'codex-plan-ceo-format-mode', - 'codex-plan-ceo-format-approach', - 'codex-plan-eng-format-coverage', - 'codex-plan-eng-format-kind', 'codex-sol-scope-termination', ].sort()); expect(selected.every((id) => E2E_TIERS[id] === 'periodic')).toBe(true); }); - test('format cases are selected by canonical source and their own test file', () => { - expect(selectedBy('plan-ceo-review/SKILL.md.tmpl')).toContain('codex-plan-ceo-format-mode'); - expect(selectedBy('plan-ceo-review/SKILL.md.tmpl')).toContain('codex-plan-ceo-format-approach'); - expect(selectedBy('plan-eng-review/SKILL.md.tmpl')).toContain('codex-plan-eng-format-coverage'); - expect(selectedBy('plan-eng-review/SKILL.md.tmpl')).toContain('codex-plan-eng-format-kind'); - expect(selectedBy('test/codex-e2e-plan-format.test.ts').length).toBe(4); - }); - test('Sol fixture generation changes select its periodic case', () => { for (const file of ['test/helpers/sol-skill-fixture.ts', 'test/sol-skill-fixture.test.ts']) { expect(selectedBy(file)).toEqual(['codex-sol-scope-termination']); diff --git a/test/conductor-prose-observation-ao.test.ts b/test/conductor-prose-observation-ao.test.ts deleted file mode 100644 index 5fb8243de..000000000 --- a/test/conductor-prose-observation-ao.test.ts +++ /dev/null @@ -1,103 +0,0 @@ -import { expect, test } from 'bun:test'; -import fs from 'node:fs'; -import path from 'node:path'; -import * as predicates from './helpers/claude-pty-runner'; -import { E2E_TOUCHFILES } from './helpers/touchfiles-data'; -import fixture from './fixtures/conductor-prose-ao.json'; -import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets'; - -const partial=fixture.publicDecisionTail; -// This next frame is synthetic; the retained live attempt ended during A. -const complete=partial+'\nB) Keep all four components and define cache invalidation before implementation.\nReply with A or B.'; -async function observe(frames:string[],verdict:'waiting'|'working',required?:boolean){ - const source=fs.readFileSync(path.join(import.meta.dir,'helpers/claude-pty-runner.ts'),'utf8'); - const start=source.indexOf('export async function runPlanSkillObservation('),end=source.indexOf('\n// ─',start); - expect(start).toBeGreaterThan(0);expect(end).toBeGreaterThan(start); - const js=new Bun.Transpiler({loader:'ts'}).transformSync(source.slice(start,end).replace('export async function','async function')+'\nreturn runPlanSkillObservation;'); - let clock=0,tick=-1,closed=0,judged=0; - const current=()=>frames[Math.min(Math.max(tick,0),frames.length-1)]!; - const args:Record={path,process:{cwd:()=>'/synthetic-owned'},Date:{now:()=>clock},randomUUID:()=> 'owned', - Bun:{sleep:async(ms:number)=>{if(ms===2000){tick++;clock+=61000;}else clock+=ms;}}, - launchClaudePty:async()=>({send:()=>{},mark:()=>0,exited:()=>false,visibleSince:current,rawOutput:current,currentScreen:async()=>current(),hermeticConfigDir:null,close:async()=>{closed++;}}), - createPlanCountSnapshotWriter:()=>()=>({}),logPtySnapshot:()=>{}, - submitPlanSeed: async () => {}, PlanSeedTimeout: class extends Error {}, - isRejectedSlashCommand:predicates.isRejectedSlashCommand, - isProseAUQVisible:predicates.isProseAUQVisible,isPlanReadyVisible:predicates.isPlanReadyVisible, - isUnknownSlashCommandVisible:predicates.isUnknownSlashCommandVisible, - isScopeGateQuestionVisible:predicates.isScopeGateQuestionVisible,isScopeGateAutoSelectVisible:predicates.isScopeGateAutoSelectVisible, - classifyVisible:predicates.classifyVisible,extractPlanFilePath:predicates.extractPlanFilePath,findNativeAutoDecision:()=>null, - judgePtyState:()=>{judged++;return {state:verdict,reasoning:'synthetic fixed verdict'};}, - }; - const run=new Function(...Object.keys(args),js)(...Object.values(args)); - const obs=await run({skillName:'plan-eng-review',initialPlanContent:'# Plan: Required draft',timeoutMs:300000,...(required===undefined?{}:{requireProseEvidence:required})}); - expect(closed).toBe(1); - return {obs,polls:tick+1,judged}; -} - -test('a judge waiting on the exact partial Conductor brief cannot stop a prose-required observation',async()=>{ - expect(fixture.actualFlags.proseAUQEverObserved).toBe(false);expect(fixture.actualFlags.waitingEverObserved).toBe(true); - expect(predicates.isProseAUQVisible(partial)).toBe(false);expect(predicates.isProseAUQVisible(complete)).toBe(true); - const {obs,polls,judged}=await observe([partial,complete],'waiting',true); - expect(polls).toBe(2);expect(judged).toBe(1);expect(obs.outcome).toBe('asked'); - expect(obs.proseAUQEverObserved).toBe(true);expect(obs.waitingEverObserved).toBe(true); -}); -test('partial-only judge waiting reaches the existing budget without gaining prose fallback credit',async()=>{ - const {obs,polls,judged}=await observe([partial],'waiting',true); - expect(polls).toBe(5);expect(judged).toBe(5);expect(obs.outcome).toBe('timeout'); - expect(obs.proseAUQEverObserved).toBe(false);expect(obs.waitingEverObserved).toBe(true); -}); -test('the prose requirement does not alter completed questions or deterministic failure precedence',async()=>{ - const done=await observe([complete],'waiting',true); - expect(done.obs.outcome).toBe('asked');expect(done.obs.proseAUQEverObserved).toBe(true);expect(done.judged).toBe(0); - const wrote=await observe(['⏺ Write(/tmp/foreign-output.md)'],'waiting',true); - expect(wrote.obs.outcome).toBe('silent_write');expect(wrote.obs.proseAUQEverObserved).toBe(false);expect(wrote.judged).toBe(0); -}); -test('other callers retain the original judge waiting behavior',async()=>{ - for(const required of [undefined,false]){ - const {obs,polls}=await observe([partial,complete],'waiting',required); - expect(polls).toBe(1);expect(obs.outcome).toBe('asked');expect(obs.proseAUQEverObserved).toBe(false);expect(obs.waitingEverObserved).toBe(true); - } - const working=await observe([partial],'working',true); - expect(working.obs.outcome).toBe('timeout');expect(working.obs.waitingEverObserved).toBe(false); -}); -async function exerciseCaller(outcome: string, proseAUQEverObserved: boolean) { - const caller=fs.readFileSync(path.join(import.meta.dir,'skill-e2e-conductor-prose.test.ts'),'utf8'); - const callbacks: Array<() => Promise> = []; - let calls = 0; - const bindings = { - expect, CAPTURE_MS, CAPTURE_LONG_MS, - describeE2ETier: (tier: string) => { - expect(tier).toBe('periodic'); - return (_title: string, register: () => void) => register(); - }, - test: (_title: string, callback: () => Promise, timeout: number) => { - expect(timeout).toBe(CAPTURE_LONG_MS); callbacks.push(callback); - }, - runPlanSkillObservation: async (opts: Record) => { - calls++; - expect(opts).toMatchObject({skillName: 'plan-eng-review', inPlanMode: true, - requireProseEvidence: true, timeoutMs: CAPTURE_MS, - env: {CONDUCTOR_WORKSPACE_PATH: '/tmp/conductor-prose-e2e'}, - extraArgs: ['--disallowedTools', 'AskUserQuestion']}); - return {outcome, proseAUQEverObserved, summary: 'controlled caller', evidence: partial}; - }, - }; - const body = caller.replace(/^import[\s\S]*?;\n/gm, ''); - new Function(...Object.keys(bindings), new Bun.Transpiler({loader: 'ts'}).transformSync(body))(...Object.values(bindings)); - expect(callbacks).toHaveLength(1); - try { await callbacks[0]!(); } - finally { expect(calls).toBe(1); } -} -test('the actual Conductor caller requests prose evidence and accepts a completed decision', async () => { - await exerciseCaller('asked', true); -}); -test.each(['asked', 'auto_decided', 'plan_ready'])('the actual Conductor caller rejects %s without independent prose evidence', async outcome => { - await expect(exerciseCaller(outcome, false)).rejects.toThrow('Conductor prose decision not observed'); -}); -test.each(['silent_write', 'timeout', 'exited'])('the actual Conductor caller preserves the %s failure even with earlier prose evidence', async outcome => { - await expect(exerciseCaller(outcome, true)).rejects.toThrow(outcome === 'silent_write' - ? 'skill wrote findings without surfacing a decision' : `outcome=${outcome}`); -}); -test('the Conductor regression and public fixture select their existing owner',()=>{ - for(const p of ['test/conductor-prose-observation-ao.test.ts','test/fixtures/conductor-prose-ao.json'])expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(p)).map(([owner])=>owner)).toEqual(['conductor-prose']); -}); diff --git a/test/cookie-validation-phases.test.ts b/test/cookie-validation-phases.test.ts index 6df841e2a..d463a7731 100644 --- a/test/cookie-validation-phases.test.ts +++ b/test/cookie-validation-phases.test.ts @@ -41,8 +41,8 @@ test('the existing quality and behavior phases retain their complete separate sh const behaviorFiles = behavior.entries.filter(entry => entry.status === 'planned').map(entry => entry.file); expect(quality.evalsAll).toBe(true); expect(behavior.evalsAll).toBe(true); - expect(qualityFiles).toHaveLength(2); - expect(behaviorFiles).toHaveLength(56); + expect(qualityFiles).toHaveLength(1); + expect(behaviorFiles).toHaveLength(51); expect(qualityFiles.every(file => file.startsWith('test/skill-llm-eval'))).toBe(true); expect(behaviorFiles.every(file => !qualityFiles.includes(file))).toBe(true); }); diff --git a/test/dx-manual-handoff-ao.test.ts b/test/dx-manual-handoff-ao.test.ts index 042bab183..0efa7fdfa 100644 --- a/test/dx-manual-handoff-ao.test.ts +++ b/test/dx-manual-handoff-ao.test.ts @@ -29,7 +29,7 @@ describe('AO completed manual DX handoff preserves report freshness',()=>{ expect(E2E_TOUCHFILES[owner]).toContain('test/fixtures/dx-manual-handoff-ao.json'); } const arrays=[...Object.values(E2E_TOUCHFILES),...Object.values(LLM_JUDGE_TOUCHFILES),GLOBAL_TOUCHFILES]; - expect(arrays).toHaveLength(244); + expect(arrays).toHaveLength(221); for(const values of arrays)for(let i=0;i{ diff --git a/test/e2e-tier-alignment.test.ts b/test/e2e-tier-alignment.test.ts index 0a95eb0ef..13f0500cb 100644 --- a/test/e2e-tier-alignment.test.ts +++ b/test/e2e-tier-alignment.test.ts @@ -57,8 +57,6 @@ const KNOWN_UNREGISTERED = new Set([ 'test/skill-e2e-auq-matrix.test.ts', // Standalone periodic self-gated A/B probe; template-literal testNames (auq-ab-${label}), no E2E map key — fail-open-safe, runs on every periodic sweep. 'test/skill-e2e-auq-verbose-vs-carved-ab.test.ts', - // bin-script pipeline test (spawns bun scripts, no model spend) that lives under the skill-e2e-* glob; no E2E map key exists for it. - 'test/skill-e2e-memory-pipeline.test.ts', ]); describe('E2E tier alignment (touchfiles declaration vs test self-gate)', () => { diff --git a/test/eng-finding-retry-budget.test.ts b/test/eng-finding-retry-budget.test.ts index fe5af917f..c9ef5bf1e 100644 --- a/test/eng-finding-retry-budget.test.ts +++ b/test/eng-finding-retry-budget.test.ts @@ -127,16 +127,16 @@ test('live periodic census fits the declared CI wall including setup', () => { return paidShardWallUpperBoundMs(files, workers); }); expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(periodicJob['timeout-minutes'] * 60_000); - expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(100); + expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(82); const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount - 1); - expect(overlays).toHaveLength(6); + expect(overlays).toHaveLength(4); expect(overlays.every(e => isOverlayTestFile(e.file))).toBe(true); expect(m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount).map(e => e.file)).toEqual([AUTOPLAN_CHAIN_BUDGET.file]); }); test('registered allocation is deterministic and preserves every discovered file', () => { const files = collectPaidTestFiles(); - expect(files).toHaveLength(119); + expect(files).toHaveLength(105); const m = livePlan(files); expect(livePlan([...files].reverse())).toEqual(m); expect(m.entries.map(e => e.file).sort()).toEqual([...files].sort()); @@ -175,7 +175,7 @@ test('current detach supervision covers the live-census floor', () => { const floor = Math.ceil((Math.ceil(files.length / DEFAULT_JOBS) * DEFAULT_SHARD_TIMEOUT_MS + excess) / 1000 * 1.05); const pkg = JSON.parse(fs.readFileSync(path.join(import.meta.dir, '../package.json'), 'utf8')); const configured = Number(pkg.scripts['eval:bg:periodic'].match(/--timeout\s+(\d+)/)[1]); - expect(floor).toBe(65541); + expect(floor).toBe(57330); expect(configured).toBeGreaterThanOrEqual(floor); expect(pkg.scripts['eval:bg:gate']).toContain('--timeout 36000'); }); diff --git a/test/eng-seeded-completion-ai.test.ts b/test/eng-seeded-completion-ai.test.ts index 43d6aa8d0..7f3995f67 100644 --- a/test/eng-seeded-completion-ai.test.ts +++ b/test/eng-seeded-completion-ai.test.ts @@ -184,7 +184,7 @@ test('rejected completion does not erase a genuine earlier question or change un test('completion evidence dependencies select exactly the seeded observation owners', () => { const owners = ['plan-ceo-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-design-review-plan-mode', - 'plan-devex-review-plan-mode', 'plan-mode-no-op', 'auto-decide-preserved', 'conductor-prose'].sort(); + 'plan-devex-review-plan-mode', 'plan-mode-no-op', 'auto-decide-preserved'].sort(); for (const file of ['test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json']) { expect(selectTests([file], E2E_TOUCHFILES).selected.sort()).toEqual(owners); } diff --git a/test/fixtures/conductor-prose-ao.json b/test/fixtures/conductor-prose-ao.json deleted file mode 100644 index 32089bbfc..000000000 --- a/test/fixtures/conductor-prose-ao.json +++ /dev/null @@ -1,24 +0,0 @@ -{ - "publicDecisionTail": "D1—ReducethepricingtiertoStripe-nativepieces,orbuildallfourcomponents?\r\rReplywithaletter(Conductorsession,sothisbriefisinprose).Project/branch:gstackedinburgh-v1,reviewingan\rexternal pricing-tier plan.\r\rELI10:Youwantmoredeveloperstosignup,soyou'readdingacheaperplan.Thedraftbuildsfournewthingstodo\rit: a Stripe price, a custm pricing web age,r own atabse tbletht remmbers who paid fowhat, ad a fast\rn-memory copy of tht table. Strpealready ships the table (Entitlements)and th page (rcing table). Every exra\rcopy of \"wh is allowed to dowhat\" is another place that can disagre withStripe when a card fails or awebhook\rarriveslate. The stakes: a developer who paidand sees \"upgrae required,\"or ono cnelled andkeeps access.\r\rStakesifwepickwrong:threesourcesoftruthforentitlementswithnoinvalidationdesignmeanssilentaccessbugs\rtht onlyshow upas angry support tickets.\r\rRecommendation:Abecauseitkeepseveryuser-visibleoutcomeofthedraftwhiledeletingthetwocomponentsmost\rlikely to causeentitlemnt drift,and it fits \"engineered enough\" an \"smallestiff that cleanly exressethe\rchange.\"\r\rCompleteness:A=9/10,B=7/10.Cisahold,notabuildoption,sonoscore.\r\rA)Stripe-nativeslimbuild(recommended).Completeness9/10.Human:~1week/CC:~1to2hours.AddoneStripePrice\rplus a Fature-to-Product mapping, mirror active_entitlement_sumryinto a singleexisting cstomer table hrough an\ridempotent webhok, gate fetures off that olumn, ship the pricing surfacebehinda feature flag (Stripe pricin\u001b[?25h", - "actualFlags": { - "proseAUQEverObserved": false, - "waitingEverObserved": true, - "scopeGateAutoSelectObserved": true - }, - "owner": { - "sessionId": "4e820ffc-b6dd-47d7-b074-5c4b0a3cbb68", - "commandStartedAt": 1789041170391, - "nativePolledAt": 1789041383968, - "captureAt": "2026-09-10T11:56:33.483Z" - }, - "provenance": { - "observation": ".context/ship-source-ao-delta-paid-20260910-v1/conductor-first-failure-ledger-v1/observation.json", - "observationSha256": "da8c8a342f44ae06bfdd092f1bdb6a395765553d65ac99eecaaa35dedb34545d", - "visible": ".context/ship-source-ao-delta-paid-20260910-v1/conductor-first-failure-ledger-v1/terminal.visible.log", - "visibleSha256": "2a2fca3e8d1b51694d76a3dbcfac524626f856e4ebf38f4bd093346edb64bf82", - "startCharacterOffset": 78469, - "publicTailSha256": "e551890b1d839cee7d7dce05120b8dc05a54f27fe8f69617aa99384c49316604", - "limit": "Exact public partial D1 tail; no completed native prose message retained. A later second option in tests is synthetic.", - "offsetUnit": "Unicode code points in decoded retained bytes, without newline normalization" - } -} diff --git a/test/fixtures/eval-baselines.json b/test/fixtures/eval-baselines.json index 1ba57b4d8..2978e45db 100644 --- a/test/fixtures/eval-baselines.json +++ b/test/fixtures/eval-baselines.json @@ -1,6 +1,4 @@ { - "command_reference": { "clarity": 4, "completeness": 3, "actionability": 4 }, - "snapshot_flags": { "clarity": 4, "completeness": 4, "actionability": 4 }, "browse_skill": { "clarity": 4, "completeness": 4, "actionability": 4 }, "qa_workflow": { "clarity": 4, "completeness": 4, "actionability": 4 }, "qa_health_rubric": { "clarity": 4, "completeness": 3, "actionability": 4 } diff --git a/test/fixtures/overlay-nudges.ts b/test/fixtures/overlay-nudges.ts index c51e61781..958193af1 100644 --- a/test/fixtures/overlay-nudges.ts +++ b/test/fixtures/overlay-nudges.ts @@ -305,50 +305,6 @@ export const OVERLAY_FIXTURES: OverlayFixture[] = [ pass: lowerIsBetter20Pct, }, - { - id: 'opus-4-7-effort-match-trivial-sonnet', - overlayPath: 'model-overlays/opus-4-7.md', - model: 'claude-sonnet-4-6', - trials: 10, - concurrency: 3, - direction: 'lower_is_better', - maxTurns: 8, - setupWorkspace: (dir) => { - fs.writeFileSync( - path.join(dir, 'config.json'), - '{"name": "demo", "version": "1.0.0"}\n', - ); - }, - userPrompt: "What's the version in config.json? Return only a JSON object with the version key and its exact string value. " + - "The final message is consumed directly by JSON.parse: return the exact JSON object, with no Markdown fences and no other prose.", - metric: reportedThinkingTokens, - metricName: 'reported_thinking_tokens', - verify: (r) => assertFinalJson(r, { version: '1.0.0' }), - comparison: { direction: 'lower_is_better', minimum: 0 }, - pass: lowerIsBetter20Pct, - }, - - { - id: 'opus-4-7-literal-interpretation-sonnet', - overlayPath: 'model-overlays/opus-4-7.md', - model: 'claude-sonnet-4-6', - trials: 10, - concurrency: 3, - direction: 'higher_is_better', - allowedTools: ['Read', 'Glob', 'Grep', 'Bash', 'Edit', 'Write'], - maxTurns: 15, - setupWorkspace: setupLiteralWorkspace, - userPrompt: 'Fix the failing tests. Preserve the specified behavior and repair the implementation.', - metric: (_r, dir, deadlineAt) => { - if (!dir) throw new Error('literal fixture metric needs its workspace'); - return correctLiteralTargets(dir, deadlineAt); - }, - metricName: 'correct_target_behaviors', - taskCorrect: (metric) => metric === 3, - allowedChanges: ['src/auth.ts', 'src/billing.ts', 'src/notifications.ts'], - comparison: { direction: 'higher_is_better', minimum: 0, maximum: 3 }, - pass: higherIsBetter20Pct, - }, ]; // Validate at module load so a broken fixture fails fast at test startup, diff --git a/test/gemini-e2e.test.ts b/test/gemini-e2e.test.ts deleted file mode 100644 index ed1270a61..000000000 --- a/test/gemini-e2e.test.ts +++ /dev/null @@ -1,174 +0,0 @@ -/** - * Gemini CLI E2E smoke test — verify Gemini CLI can start and discover skills. - * - * This is a lightweight smoke test, not a full integration test. Gemini CLI - * gets lost in worktrees and times out on complex tasks. The smoke test - * validates that the skill files are structured correctly for Gemini's - * .agents/skills/ discovery mechanism. - * - * Prerequisites: - * - `gemini` binary installed (npm install -g @google/gemini-cli) - * - Gemini authenticated via ~/.gemini/ config or GEMINI_API_KEY env var - * - EVALS=1 env var set (same gate as Claude E2E tests) - * - * Skips gracefully when prerequisites are not met. - */ - -import { describe, test, expect, beforeAll, afterAll } from 'bun:test'; -import { JUDGE_MS } from './helpers/eval-budgets'; -import { runGeminiSkill } from './helpers/gemini-session-runner'; -import type { GeminiResult } from './helpers/gemini-session-runner'; -import { EvalCollector } from './helpers/eval-store'; -import { selectTests, detectBaseBranch, getChangedFiles, E2E_TOUCHFILES, GLOBAL_TOUCHFILES } from './helpers/touchfiles'; -import { createTestWorktree, harvestAndCleanup } from './helpers/e2e-helpers'; -import * as path from 'path'; - -const ROOT = path.resolve(import.meta.dir, '..'); - -// --- Prerequisites check --- - -const GEMINI_AVAILABLE = (() => { - try { - const result = Bun.spawnSync(['which', 'gemini'], { timeout: 30_000 }); - return result.exitCode === 0; - } catch { return false; } -})(); - -// A binary on PATH is not enough: the CLI can be present but UNUSABLE — the -// individual code-assist auth path was deprecated upstream ("migrate to the -// Antigravity suite"), which fails every run before any model call, and flag -// churn (--skip-trust removed in 0.34) errors at argv parse. Probe with a -// bare --help: a CLI that can't even print usage is unusable, and a working -// one is cheap to confirm. Deeper auth failures are classified per-run. -const GEMINI_USABLE = GEMINI_AVAILABLE && (() => { - try { - const result = Bun.spawnSync(['gemini', '--help'], { timeout: 15_000 }); - return result.exitCode === 0; - } catch { return false; } -})(); - -const evalsEnabled = !!process.env.EVALS; - -// External-service tests are periodic-tier (CLAUDE.md tiering rule 3): -// "Requires external service (Codex, Gemini)? -> periodic". The positive -// form below is the canonical whole-file guard shape — the sharded runner's -// classifyPaidTestFile greps for it to exclude this file from gate. -const tierOk = process.env.EVALS_TIER === 'periodic'; - -// Skip all tests if gemini is not available/usable, EVALS is not set, or -// we're in the gate tier. -const SKIP = !GEMINI_USABLE || !evalsEnabled || !tierOk; - -const describeGemini = SKIP ? describe.skip : describe; - -// Log why we're skipping (helpful for debugging CI) -if (!evalsEnabled) { - // Silent — same as Claude E2E tests, EVALS=1 required -} else if (!tierOk) { - process.stderr.write('\nGemini E2E: SKIPPED — external-service test, periodic tier only (EVALS_TIER === \'periodic\')\n'); -} else if (!GEMINI_AVAILABLE) { - process.stderr.write('\nGemini E2E: SKIPPED — gemini binary not found (install: npm i -g @google/gemini-cli)\n'); -} else if (!GEMINI_USABLE) { - process.stderr.write('\nGemini E2E: SKIPPED — gemini CLI present but unusable (auth path deprecated upstream or CLI broken; try updating @google/gemini-cli)\n'); -} - -// --- Diff-based test selection --- - -// Gemini E2E touchfiles — DERIVED from the canonical map, never a local fork -// (the old hand-copy kept a gitignored '.agents/skills/**' pattern that can -// never match a git diff and missed canonical deps — same drift class as the -// codex copy). -const GEMINI_E2E_TOUCHFILES: Record = Object.fromEntries( - (['gemini-smoke'] as const).map((key) => { - if (!E2E_TOUCHFILES[key]) throw new Error(`canonical E2E_TOUCHFILES lost key '${key}' — fix the map, not this file`); - return [key, E2E_TOUCHFILES[key]]; - }), -); - -let selectedTests: string[] | null = null; // null = run all - -if (evalsEnabled && !process.env.EVALS_ALL) { - const baseBranch = process.env.EVALS_BASE - || detectBaseBranch(ROOT) - || 'main'; - const changedFiles = getChangedFiles(baseBranch, ROOT); - - if (changedFiles.length > 0) { - const selection = selectTests(changedFiles, GEMINI_E2E_TOUCHFILES, GLOBAL_TOUCHFILES); - selectedTests = selection.selected; - process.stderr.write(`\nGemini E2E selection (${selection.reason}): ${selection.selected.length}/${Object.keys(GEMINI_E2E_TOUCHFILES).length} tests\n`); - if (selection.skipped.length > 0) { - process.stderr.write(` Skipped: ${selection.skipped.join(', ')}\n`); - } - process.stderr.write('\n'); - } -} - -/** Skip an individual test if not selected by diff-based selection. */ -function testIfSelected(testName: string, fn: () => Promise, timeout: number) { - const shouldRun = selectedTests === null || selectedTests.includes(testName); - (shouldRun ? test.concurrent : test.skip)(testName, fn, timeout); -} - -// --- Eval result collector --- - -const evalCollector = evalsEnabled && !SKIP ? new EvalCollector('e2e-gemini') : null; - -function recordGeminiE2E(name: string, result: GeminiResult, passed: boolean) { - evalCollector?.addTest({ - name, - suite: 'gemini-e2e', - tier: 'e2e', - passed, - duration_ms: result.durationMs, - cost_usd: 0, - output: result.output?.slice(0, 2000), - turns_used: result.toolCalls.length, - exit_reason: result.exitCode === 0 ? 'success' : `exit_code_${result.exitCode}`, - }); -} - -function logGeminiCost(label: string, result: GeminiResult) { - const durationSec = Math.round(result.durationMs / 1000); - console.log(`${label}: ${result.tokens} tokens, ${result.toolCalls.length} tool calls, ${durationSec}s`); -} - -// Finalize eval results on exit -afterAll(async () => { - if (evalCollector) { - await evalCollector.finalize(); - } -}); - -// --- Tests --- - -describeGemini('Gemini E2E', () => { - let testWorktree: string; - - beforeAll(() => { - testWorktree = createTestWorktree('gemini'); - }); - - afterAll(() => { - harvestAndCleanup('gemini'); - }); - - testIfSelected('gemini-smoke', async () => { - // Smoke test: can Gemini start, read the repo, and produce output? - // Uses a simple prompt that doesn't require skill invocation or complex navigation. - const result = await runGeminiSkill({ - prompt: 'What is this project? Answer in one sentence based on the README.', - timeoutMs: JUDGE_MS, - cwd: testWorktree, - }); - - logGeminiCost('gemini-smoke', result); - - // Pass if Gemini produced any meaningful output (even with non-zero exit from timeout) - const hasOutput = result.output.length > 10; - const passed = hasOutput; - recordGeminiE2E('gemini-smoke', result, passed); - - expect(result.output.length, 'Gemini should produce output').toBeGreaterThan(10); - }, JUDGE_MS); -}); diff --git a/test/gstack-skill-start.test.ts b/test/gstack-skill-start.test.ts index 28e75fca9..f2b133ab4 100644 --- a/test/gstack-skill-start.test.ts +++ b/test/gstack-skill-start.test.ts @@ -369,6 +369,40 @@ describe('gstack-skill-start behavior', () => { const out = runStart(); expect(out).toMatch(/^ARTIFACTS_SYNC: off$/m); }); + + test('artifacts-sync consent is asked before any artifacts egress, and only in interactive sessions', () => { + const gh = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ss-privacy-')); + const bin = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ss-privacy-bin-')); + const remote = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ss-privacy-remote-')); + const git = (args: string[], cwd: string) => execFileSync('git', args, { cwd, timeout: 30_000, stdio: 'pipe' }); + try { + fs.writeFileSync(path.join(bin, 'gbrain'), '#!/bin/sh\nexit 0\n', { mode: 0o755 }); + git(['init', '-q', '--bare'], remote); + git(['init', '-q', '-b', 'main'], gh); + git(['remote', 'add', 'origin', remote], gh); + const pullStamp = path.join(gh, '.brain-last-pull'); + const start = (config: string, env: Record = {}) => { + fs.writeFileSync(path.join(gh, 'config.yaml'), `update_check: false\n${config}`); + return runStart([], { GSTACK_HOME: gh, PATH: `${bin}${path.delimiter}${process.env.PATH}`, ...env }); + }; + const gates = (out: string) => (out.match(/^GSTACK_INSTRUCTION_BEGIN: privacy-stop-gate/gm) ?? []).length; + + const pending = start(''); + expect(gates(pending)).toBe(1); + expect(pending).toContain('How much should sync?'); + expect(pending).toMatch(/^ARTIFACTS_SYNC: off$/m); + expect(fs.existsSync(pullStamp)).toBe(false); + + expect(gates(start('', { GSTACK_SESSION_KIND: 'spawned' }))).toBe(0); + expect(fs.existsSync(pullStamp)).toBe(false); + + const consented = start('artifacts_sync_mode: full\nartifacts_sync_mode_prompted: true\n'); + expect(gates(consented)).toBe(0); + expect(fs.existsSync(pullStamp)).toBe(true); + } finally { + for (const dir of [gh, bin, remote]) fs.rmSync(dir, { recursive: true, force: true }); + } + }); }); describe('gstack-skill-end', () => { diff --git a/test/helpers/eval-budgets.ts b/test/helpers/eval-budgets.ts index 6ea57772d..680364970 100644 --- a/test/helpers/eval-budgets.ts +++ b/test/helpers/eval-budgets.ts @@ -106,9 +106,8 @@ export const FILE_RETRY_BUDGETS = [ ...STRICT_RETRY_CASE_BUDGETS, ...[ // Sixteen workflow judges include their 10s recording grace; the other - // eleven judges retain 120s. Supervise all 27 and the existing one retry. - { file: 'test/skill-llm-eval.test.ts', attemptMs: 16 * (JUDGE_MS + 10_000) + 11 * JUDGE_MS, retries: 1 }, - { file: 'test/codex-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_LONG_MS + 10_000), retries: 1 }, + // seven judges retain 120s. Supervise all 23 and the existing one retry. + { file: 'test/skill-llm-eval.test.ts', attemptMs: 16 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, retries: 1 }, { file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 }, { file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 }, { file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, retries: 1 }, diff --git a/test/helpers/gemini-session-runner.test.ts b/test/helpers/gemini-session-runner.test.ts deleted file mode 100644 index 1bb9a3939..000000000 --- a/test/helpers/gemini-session-runner.test.ts +++ /dev/null @@ -1,104 +0,0 @@ -import { describe, test, expect } from 'bun:test'; -import { parseGeminiJSONL } from './gemini-session-runner'; - -// Fixture: actual Gemini CLI stream-json output with tool use -const FIXTURE_LINES = [ - '{"type":"init","timestamp":"2026-03-20T15:14:46.455Z","session_id":"test-session-123","model":"auto-gemini-3"}', - '{"type":"message","timestamp":"2026-03-20T15:14:46.456Z","role":"user","content":"list the files"}', - '{"type":"message","timestamp":"2026-03-20T15:14:49.650Z","role":"assistant","content":"I will list the files.","delta":true}', - '{"type":"tool_use","timestamp":"2026-03-20T15:14:49.690Z","tool_name":"run_shell_command","tool_id":"cmd_1","parameters":{"command":"ls"}}', - '{"type":"tool_result","timestamp":"2026-03-20T15:14:49.931Z","tool_id":"cmd_1","status":"success","output":"file1.ts\\nfile2.ts"}', - '{"type":"message","timestamp":"2026-03-20T15:14:51.945Z","role":"assistant","content":"Here are the files.","delta":true}', - '{"type":"result","timestamp":"2026-03-20T15:14:52.030Z","status":"success","stats":{"total_tokens":27147,"input_tokens":26928,"output_tokens":87,"cached":0,"duration_ms":5575,"tool_calls":1}}', -]; - -describe('parseGeminiJSONL', () => { - test('extracts session ID from init event', () => { - const parsed = parseGeminiJSONL(FIXTURE_LINES); - expect(parsed.sessionId).toBe('test-session-123'); - }); - - test('concatenates assistant message deltas into output', () => { - const parsed = parseGeminiJSONL(FIXTURE_LINES); - expect(parsed.output).toBe('I will list the files.Here are the files.'); - }); - - test('ignores user messages', () => { - const lines = [ - '{"type":"message","role":"user","content":"this should be ignored"}', - '{"type":"message","role":"assistant","content":"this should be kept","delta":true}', - ]; - const parsed = parseGeminiJSONL(lines); - expect(parsed.output).toBe('this should be kept'); - }); - - test('extracts tool names from tool_use events', () => { - const parsed = parseGeminiJSONL(FIXTURE_LINES); - expect(parsed.toolCalls).toHaveLength(1); - expect(parsed.toolCalls[0]).toBe('run_shell_command'); - }); - - test('extracts total tokens from result stats', () => { - const parsed = parseGeminiJSONL(FIXTURE_LINES); - expect(parsed.tokens).toBe(27147); - }); - - test('skips malformed lines without throwing', () => { - const lines = [ - '{"type":"init","session_id":"ok"}', - 'this is not json', - '{"type":"message","role":"assistant","content":"hello","delta":true}', - '{incomplete json', - '{"type":"result","status":"success","stats":{"total_tokens":100}}', - ]; - const parsed = parseGeminiJSONL(lines); - expect(parsed.sessionId).toBe('ok'); - expect(parsed.output).toBe('hello'); - expect(parsed.tokens).toBe(100); - }); - - test('skips empty and whitespace-only lines', () => { - const lines = [ - '', - ' ', - '{"type":"init","session_id":"s1"}', - '\t', - '{"type":"result","status":"success","stats":{"total_tokens":50}}', - ]; - const parsed = parseGeminiJSONL(lines); - expect(parsed.sessionId).toBe('s1'); - expect(parsed.tokens).toBe(50); - }); - - test('handles empty input', () => { - const parsed = parseGeminiJSONL([]); - expect(parsed.output).toBe(''); - expect(parsed.toolCalls).toHaveLength(0); - expect(parsed.tokens).toBe(0); - expect(parsed.sessionId).toBeNull(); - }); - - test('handles missing fields gracefully', () => { - const lines = [ - '{"type":"init"}', // no session_id - '{"type":"message","role":"assistant"}', // no content - '{"type":"tool_use"}', // no tool_name - '{"type":"result","status":"success"}', // no stats - ]; - const parsed = parseGeminiJSONL(lines); - expect(parsed.sessionId).toBeNull(); - expect(parsed.output).toBe(''); - expect(parsed.toolCalls).toHaveLength(0); - expect(parsed.tokens).toBe(0); - }); - - test('handles multiple tool_use events', () => { - const lines = [ - '{"type":"tool_use","tool_name":"run_shell_command","tool_id":"cmd_1","parameters":{"command":"ls"}}', - '{"type":"tool_use","tool_name":"read_file","tool_id":"cmd_2","parameters":{"path":"foo.ts"}}', - '{"type":"tool_use","tool_name":"run_shell_command","tool_id":"cmd_3","parameters":{"command":"cat bar.ts"}}', - ]; - const parsed = parseGeminiJSONL(lines); - expect(parsed.toolCalls).toEqual(['run_shell_command', 'read_file', 'run_shell_command']); - }); -}); diff --git a/test/helpers/gemini-session-runner.ts b/test/helpers/gemini-session-runner.ts deleted file mode 100644 index e42f53ade..000000000 --- a/test/helpers/gemini-session-runner.ts +++ /dev/null @@ -1,262 +0,0 @@ -/** - * Gemini CLI subprocess runner for skill E2E testing. - * - * Spawns `gemini -p` as an independent process, parses its stream-json - * output, and returns structured results. Follows the same pattern as - * codex-session-runner.ts but adapted for the Gemini CLI. - * - * Key differences from Codex session-runner: - * - Uses `gemini -p` instead of `codex exec` - * - Output is NDJSON with event types: init, message, tool_use, tool_result, result - * - Uses `--output-format stream-json --yolo` instead of `--json -s read-only` - * (`--skip-trust` was removed in gemini-cli 0.34; folder trust is settings-driven now) - * - No temp HOME needed — Gemini discovers skills from `.agents/skills/` in cwd - * - Message events are streamed with `delta: true` — must concatenate - */ - -import * as path from 'path'; -import { spawn } from 'child_process'; -import { Readable } from 'node:stream'; -import { hermeticChildEnv } from './hermetic-env'; -import { killProcessGroup } from '../../scripts/test-strict-output'; - -// --- Interfaces --- - -export interface GeminiResult { - output: string; // Full assistant message text (concatenated deltas) - toolCalls: string[]; // Tool names from tool_use events - tokens: number; // Total tokens used - exitCode: number; // Process exit code - durationMs: number; // Wall clock time - sessionId: string | null; // Session ID from init event - rawLines: string[]; // Raw JSONL lines for debugging -} - -// --- JSONL parser --- - -export interface ParsedGeminiJSONL { - output: string; - toolCalls: string[]; - tokens: number; - sessionId: string | null; -} - -/** - * Parse an array of JSONL lines from `gemini -p --output-format stream-json`. - * Pure function — no I/O, no side effects. - * - * Handles these Gemini event types: - * - init → extract session_id - * - message (role=assistant, delta=true) → concatenate content into output - * - tool_use → extract tool_name - * - tool_result → logged but not extracted - * - result → extract token usage from stats - */ -export function parseGeminiJSONL(lines: string[]): ParsedGeminiJSONL { - const outputParts: string[] = []; - const toolCalls: string[] = []; - let tokens = 0; - let sessionId: string | null = null; - - for (const line of lines) { - if (!line.trim()) continue; - try { - const obj = JSON.parse(line); - const t = obj.type || ''; - - if (t === 'init') { - const sid = obj.session_id || ''; - if (sid) sessionId = sid; - } else if (t === 'message') { - if (obj.role === 'assistant' && obj.content) { - outputParts.push(obj.content); - } - } else if (t === 'tool_use') { - const name = obj.tool_name || ''; - if (name) toolCalls.push(name); - } else if (t === 'result') { - const stats = obj.stats || {}; - tokens = (stats.total_tokens || 0); - } - } catch { /* skip malformed lines */ } - } - - return { - output: outputParts.join(''), - toolCalls, - tokens, - sessionId, - }; -} - -// --- Main runner --- - -/** - * Run a prompt via `gemini -p` and return structured results. - * - * Spawns gemini with stream-json output, parses JSONL events, - * and returns a GeminiResult. Skips gracefully if gemini binary is not found. - */ -export async function runGeminiSkill(opts: { - prompt: string; // What to ask Gemini - timeoutMs?: number; // Default 300000 (5 min) - cwd?: string; // Working directory (where .agents/skills/ lives) -}): Promise { - const { - prompt, - timeoutMs = 300_000, - cwd, - } = opts; - - const startTime = Date.now(); - - // Check if gemini binary exists - const whichResult = Bun.spawnSync(['which', 'gemini'], { timeout: 30_000 }); - if (whichResult.exitCode !== 0) { - return { - output: 'SKIP: gemini binary not found', - toolCalls: [], - tokens: 0, - exitCode: -1, - durationMs: Date.now() - startTime, - sessionId: null, - rawLines: [], - }; - } - - // Build gemini command. - // --skip-trust was REMOVED in gemini-cli 0.34 ("Unknown arguments: - // skip-trust"); folder trust moved to settings and no longer needs a flag - // for headless runs. --yolo still auto-approves tool actions. - const args = ['-p', prompt, '--output-format', 'stream-json', '--yolo']; - - // Spawn gemini — uses real HOME for auth (~/.gemini; HOME is allowlisted), - // cwd for skill discovery. Hermetic scrub with gemini's auth surface - // re-admitted (previously this spawn inherited the full operator env). - // node:child_process spawn with `detached` (own process group) — mirrors - // session-runner.ts. A bare kill signalled only gemini itself; tool - // subprocesses survived as orphans holding our pipes open (the same - // blocked-drain hang the claude runner fixed — this copy lacked it). - const proc = spawn('gemini', args, { - cwd: cwd || process.cwd(), - stdio: ['ignore', 'pipe', 'pipe'], - detached: process.platform !== 'win32', - env: hermeticChildEnv(undefined, { - extraAllow: ['GEMINI_API_KEY', 'GOOGLE_API_KEY', 'GOOGLE_APPLICATION_CREDENTIALS', 'GOOGLE_CLOUD_*', 'GEMINI_*'], - }), - }); - const stdoutWeb = Readable.toWeb(proc.stdout!) as ReadableStream; - const stderrWeb = Readable.toWeb(proc.stderr!) as ReadableStream; - const procExited: Promise = new Promise((resolve) => { - proc.on('close', (code) => resolve(code ?? 1)); - proc.on('error', () => resolve(1)); - }); - - // Race against timeout - let timedOut = false; - const timeoutId = setTimeout(() => { - timedOut = true; - // Group SIGKILL + reader cancel: kill the whole tree AND unblock the - // read loop even if a stray grandchild survives the group kill. - killProcessGroup(proc, 'SIGKILL'); - reader.cancel().catch(() => { /* stream already closed */ }); - }, timeoutMs); - - // Stream and collect JSONL from stdout - const collectedLines: string[] = []; - const stderrPromise = new Response(stderrWeb).text(); - - const reader = stdoutWeb.getReader(); - const decoder = new TextDecoder(); - let buf = ''; - - try { - while (true) { - const { done, value } = await reader.read(); - if (done) break; - buf += decoder.decode(value, { stream: true }); - const lines = buf.split('\n'); - buf = lines.pop() || ''; - for (const line of lines) { - if (!line.trim()) continue; - collectedLines.push(line); - - // Real-time progress to stderr - try { - const event = JSON.parse(line); - if (event.type === 'tool_use' && event.tool_name) { - const elapsed = Math.round((Date.now() - startTime) / 1000); - process.stderr.write(` [gemini ${elapsed}s] tool: ${event.tool_name}\n`); - } else if (event.type === 'message' && event.role === 'assistant' && event.content) { - const elapsed = Math.round((Date.now() - startTime) / 1000); - process.stderr.write(` [gemini ${elapsed}s] message: ${event.content.slice(0, 100)}\n`); - } - } catch { /* skip — parseGeminiJSONL will handle it later */ } - } - } - } catch { /* stream read error — fall through to exit code handling */ } - - // Flush remaining buffer - if (buf.trim()) { - collectedLines.push(buf); - } - - // Same orphan hazard as stdout: a grandchild holding stderr open would - // block this drain forever. Race against child exit + a short grace window - // (ported from session-runner.ts — the gemini copy lacked it). - const stderr = await Promise.race([ - stderrPromise, - (async () => { - await procExited; - await new Promise((r) => setTimeout(r, 5_000)); - return ''; - })(), - ]); - const exitCode = await procExited; - clearTimeout(timeoutId); - - const durationMs = Date.now() - startTime; - - // Parse all collected JSONL lines - const parsed = parseGeminiJSONL(collectedLines); - - // Log stderr if non-empty (may contain auth errors, etc.) - if (stderr.trim()) { - process.stderr.write(` [gemini stderr] ${stderr.trim().slice(0, 200)}\n`); - } - - // Environment-unusable classification: these are Google-side conditions no - // test assertion can act on — the deprecated individual code-assist auth - // path ("migrate to the Antigravity suite") and argv drift on older/newer - // CLIs. Return the same SKIP shape as binary-not-found so callers report - // SKIPPED instead of a false FAIL. - const unusableMarkers = [ - 'no longer supported for Gemini Code Assist', - 'antigravity', - 'Unknown arguments: skip-trust', - ]; - if (exitCode !== 0 && parsed.tokens === 0) { - const marker = unusableMarkers.find((m) => stderr.toLowerCase().includes(m.toLowerCase())); - if (marker) { - return { - output: `SKIP: gemini CLI unusable (${marker})`, - toolCalls: [], - tokens: 0, - exitCode: -1, - durationMs, - sessionId: null, - rawLines: collectedLines, - }; - } - } - - return { - output: parsed.output, - toolCalls: parsed.toolCalls, - tokens: parsed.tokens, - exitCode: timedOut ? 124 : exitCode, - durationMs, - sessionId: parsed.sessionId, - rawLines: collectedLines, - }; -} diff --git a/test/helpers/overlay-case-policy.ts b/test/helpers/overlay-case-policy.ts index d703544e8..454568ccf 100644 --- a/test/helpers/overlay-case-policy.ts +++ b/test/helpers/overlay-case-policy.ts @@ -21,6 +21,4 @@ export const OVERLAY_CASE_FILES: Record = { 'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts': 'opus-4-7-effort-match-trivial', 'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts': 'opus-4-7-literal-interpretation', 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts': 'claude-dedicated-tools-vs-bash-sonnet', - 'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial-sonnet.test.ts': 'opus-4-7-effort-match-trivial-sonnet', - 'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation-sonnet.test.ts': 'opus-4-7-literal-interpretation-sonnet', }; diff --git a/test/helpers/paid-test-set.ts b/test/helpers/paid-test-set.ts index 1e317e095..94cd5e0c9 100644 --- a/test/helpers/paid-test-set.ts +++ b/test/helpers/paid-test-set.ts @@ -11,8 +11,8 @@ import { matchGlob } from './touchfiles'; /** The exact globs package.json's `test:gate` passes to `bun test`. */ export const PAID_TEST_GLOBS = [ - // skill-llm-eval* (not just the base file): skill-llm-eval-spec.test.ts - // fell outside the exact glob and could never run in any lane. + // skill-llm-eval* (not just the base file): a sibling judge file outside + // the exact glob could never run in any lane. 'test/skill-llm-eval*.test.ts', 'test/skill-e2e-*.test.ts', 'test/skill-routing-e2e.test.ts', @@ -22,7 +22,6 @@ export const PAID_TEST_GLOBS = [ // entered the paid census. The same bug class as the deleted pre-split // monolith (see test/paid-shards.test.ts's regression pin). 'test/codex-e2e*.test.ts', - 'test/gemini-e2e.test.ts', 'test/llm-judge-recommendation.test.ts', 'test/carve-section-loading*.test.ts', ] as const; diff --git a/test/helpers/periodic-exclude-data.ts b/test/helpers/periodic-exclude-data.ts index 295c177cf..a3f7ccb31 100644 --- a/test/helpers/periodic-exclude-data.ts +++ b/test/helpers/periodic-exclude-data.ts @@ -15,20 +15,32 @@ * the re-entry mechanism. */ export const PERIODIC_CI_EXCLUDE: Record = { - 'test/skill-e2e-ship-idempotency.test.ts': { - reason: - 'documented-red: the PTY child sits at the Claude Code welcome screen for the full budget ' - + '(readiness/typing race vs CLI 2.1.x); never green since it was born in v1.63', - tracking: 'TODOS.md "periodic tier — three documented-red tests need structural repair" (1 of 3 resolved: sidebar trio already deleted)', + 'test/codex-e2e.test.ts': { + reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips', + tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)', }, - 'test/skill-e2e-brain-privacy-gate.test.ts': { - reason: - 'documented-red: the artifacts-sync stop-gate preconditions do not survive the hermetic env ' - + 'even with per-test HOME/GSTACK_HOME injection; never green anywhere', - tracking: 'TODOS.md "periodic tier — three documented-red tests need structural repair"', + 'test/codex-e2e-sol-scope.test.ts': { + reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips', + tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)', }, - 'test/skill-e2e-ios.test.ts': { - reason: 'requires a live iOS device/simulator toolchain (xcodebuild, devicectl) — manual hardware, not a CI runner capability', - tracking: 'TODOS.md "skill-e2e-ios CI story" (device/runner decision)', + 'test/codex-e2e-shared-libs.test.ts': { + reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips', + tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)', + }, + 'test/codex-e2e-recommendation-substance.test.ts': { + reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips', + tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)', + }, + 'test/skill-e2e-outside-voice.test.ts': { + reason: 'needs both the claude and codex CLIs; codex is not in the CI image, so every case self-skips', + tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)', + }, + 'test/skill-e2e-aside.test.ts': { + reason: 'needs macOS with the Aside app open (asideAvailable()); CI runners are Linux, so every case self-skips', + tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)', + }, + 'test/skill-e2e-ios-device.test.ts': { + reason: 'needs a physical iPhone over USB/devicectl — manual hardware, not a CI runner capability', + tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)', }, }; diff --git a/test/helpers/providers/types.ts b/test/helpers/providers/types.ts index 827adf94b..59063e7b2 100644 --- a/test/helpers/providers/types.ts +++ b/test/helpers/providers/types.ts @@ -3,8 +3,8 @@ import * as path from 'node:path'; /** * Provider adapter interface — uniform contract for Claude, GPT, Gemini. * - * Each adapter wraps an existing runner (session-runner.ts, codex-session-runner.ts, - * gemini-session-runner.ts) and normalizes its per-provider result shape into the + * Each adapter wraps an existing runner or CLI (session-runner.ts, + * codex-session-runner.ts, the gemini CLI) and normalizes its per-provider result shape into the * RunResult below. The benchmark harness only talks to adapters through this * interface, never to the underlying runners directly. */ diff --git a/test/helpers/touchfiles-data.ts b/test/helpers/touchfiles-data.ts index bc2f30f97..a9a29123a 100644 --- a/test/helpers/touchfiles-data.ts +++ b/test/helpers/touchfiles-data.ts @@ -334,23 +334,6 @@ export const E2E_TOUCHFILES: Record = { // Conductor → prose decision brief (Conductor signal makes prose the default; // the PreToolUse hook denies the flaky tool). Touches the resolver that owns // the Conductor rule, the preamble signal, the hook, and the detection helper. - 'conductor-prose': [ - 'lib/claude-public-transcript.ts', - 'test/auto-decide-recommendation-scope.test.ts', - 'test/fixtures/auto-decide-recommendation-361c.json', - 'test/auto-decide-target-identity.test.ts', - 'test/fixtures/auto-decide-target-361c.json', - - 'test/autoplan-public-narration.test.ts', 'test/fixtures/autoplan-public-narration-ad.json', - 'test/helpers/plan-count-transcript.ts', 'test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures/plan-count-cross-cwd-ancestry-0bcd.json', 'test/plan-count-session-cwd.test.ts', - 'scripts/resolvers/learnings.ts', - "test/plan-scope-recovery-av.test.ts", - "test/fixtures/plan-scope-recovery-av.json", 'test/pty-screen-unicode-ap.test.ts', 'test/eng-scope-entry-ap.test.ts', - 'test/conductor-prose-observation-ao.test.ts', 'test/fixtures/conductor-prose-ao.json', - 'test/auto-decide-saved-ai.test.ts', 'test/fixtures/auto-decide-saved-ai.json', 'test/fixtures/auto-decide-retry-ai.json','bin/gstack-skill-start', 'bin/gstack-skill-end', 'bin/gstack-session-kind', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-preamble-bash.ts', 'scripts/resolvers/preamble.ts', 'plan-eng-review/**', 'hosts/claude/hooks/question-preference-hook.ts', 'hosts/claude/hooks/spawned-directive.ts', 'lib/is-conductor.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-conductor-prose.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', 'test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', 'test/fixtures/auto-decide-current-declaration-6aef.json', 'test/auto-decide-explanatory-mode.test.ts', 'test/fixtures/auto-decide-explanatory-mode-043a.json', 'test/fixtures/auto-decide-explanatory-mode-749df.json', 'test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'test/native-auto-decide.test.ts', 'test/native-auto-decide-pty.test.ts', 'test/helpers/fake-plan-seed.ts', 'test/fixtures/native-auto-decide-ag.json', 'test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json', 'test/helpers/plan-count-pending-exit.ts', 'test/plan-count-pending-exit.test.ts', 'test/helpers/pty-screen.ts', 'test/pty-screen.test.ts', 'test/pty-screen-session.test.ts', 'test/fixtures/pty-screen/**', - "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-completion-status.ts", - 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts', 'test/helpers/owned-claude-transcript.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/testing.ts', 'scripts/resolvers/review.ts', 'test/plan-review-cases.test.ts' - ], // Native question capture and interactive workflow probes. 'auq-format-gate': ['test/session-runner-stream-lifecycle.test.ts', 'plan-ceo-review/**', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'scripts/resolvers/preamble/generate-completeness-section.ts', 'scripts/resolvers/preamble.ts', 'test/helpers/auq-sdk-capture.ts', 'test/helpers/session-runner.ts', 'test/helpers/llm-judge.ts', 'test/skill-e2e-ask-user-question-format-compliance.test.ts', @@ -393,9 +376,6 @@ export const E2E_TOUCHFILES: Record = { "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/plan-design-with-ui-fixture.test.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/helpers/plan-review-cases.ts', 'test/helpers/plan-review-board-feedback.ts', 'test/plan-review-board-feedback.test.ts', 'test/fixtures/design-board-questions.json', 'test/fixtures/design-outside-voices-question.json', 'design/src/daemon-state.ts', 'design/src/daemon.ts', 'design/test/daemon-tests-fixtures.ts', 'design/src/daemon-client.ts', 'test/helpers/owned-claude-transcript.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'bin/gstack-paths', 'bin/gstack-slug', 'scripts/resolvers/design.ts' ], - 'ship-idempotency-pty': ['ship/**', 'bin/gstack-next-version', 'bin/gstack-version-bump', 'scripts/resolvers/sections.ts', 'lib/worktree.ts', 'test/helpers/claude-pty-runner.ts', 'test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/helpers/hermetic-skill-runtime.ts', 'test/hermetic-skill-runtime.test.ts', 'test/helpers/pty-trust-dialog.ts', 'test/pty-trust-dialog.test.ts', 'test/skill-e2e-ship-idempotency.test.ts', 'test/plan-count-truncated-border.test.ts', 'test/fixtures/eng-d2-truncated-border-0bcd.json', 'test/plan-count-truncated-question.test.ts', 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', 'test/fixtures/ceo-approach-z-call.json', 'test/fixtures/ceo-approach-z-screen.txt', - 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'scripts/resolvers/testing.ts' - ], 'tpa-present': ['test/session-runner-stream-lifecycle.test.ts', 'scripts/resolvers/third-party-actions.ts', 'ship/SKILL.md.tmpl', 'ship/sections/apple-release.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/helpers/session-runner.ts', 'test/skill-e2e-third-party-actions.test.ts', 'test/helpers/third-party-actions.ts', 'test/third-party-actions-recording.test.ts', 'lib/eval-model.ts' ], @@ -784,9 +764,6 @@ export const E2E_TOUCHFILES: Record = { 'test/fixtures/plan-count-quoted-frame-ak.json', 'docs/askuserquestion-split.md', 'test/resolver-ask-user-format.test.ts', 'bin/gstack-slug', 'test/helpers/hermetic-env.test.ts', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'lib/fs-atomic.ts', 'test/helpers/ceo-finding-fixture.ts', 'test/ceo-finding-fixture.test.ts', 'test/helpers/owned-claude-transcript.ts', 'test/eval-budgets-policy.test.ts', 'test/fixtures/webfetch-permission.json', 'test/plan-skill-webfetch-permission.test.ts', 'test/helpers/plan-skill-questions.ts', 'test/fixtures/eng-auq-validation-error.json', 'test/fixtures/bash-directory-permission.json', 'test/fixtures/design-tasks-bash-permission.json', 'test/plan-skill-read-permission.test.ts', 'test/fixtures/read-permission.json', 'test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json', 'test/plan-skill-questions.test.ts', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', 'test/helpers/plan-skill-question-hook-scope.ts', 'test/helpers/skill-census.ts', 'test/plan-skill-question-hook-scope.test.ts', 'test/helpers/plan-review-decisions.ts', 'test/plan-review-decisions.test.ts', 'test/helpers/plan-review-cases.ts', 'test/plan-review-cases.test.ts', 'test/helpers/llm-judge.ts', 'lib/eval-model.ts', 'test/skill-e2e-plan-decision-classification.test.ts', 'test/fixtures/plan-decision-classification.ts', 'test/plan-review-calibration.test.ts', 'test/helpers/ceo-split-question-policy.ts', 'test/fixtures/ceo-split-actor-6aef.json', 'test/helpers/ceo-mode-option.ts', 'test/ceo-mode-option.test.ts', 'test/ceo-split-collection.test.ts', 'test/fixtures/ceo-split-collection-0bcd.json', 'test/ceo-split-question-policy.test.ts', 'scripts/resolvers/review.ts', 'scripts/resolvers/tasks-section.ts' ], - 'brain-privacy-gate': ['bin/gstack-skill-start', 'bin/gstack-skill-end', 'scripts/resolvers/preamble/generate-brain-sync-block.ts', 'scripts/resolvers/preamble.ts', 'bin/gstack-brain-sync', 'bin/gstack-artifacts-init', 'bin/gstack-config', 'test/helpers/agent-sdk-runner.ts', 'test/skill-e2e-brain-privacy-gate.test.ts', - 'test/agent-sdk-runner.test.ts' - ], // /setup-gbrain Path 4 (Remote MCP) — happy + bad-token end-to-end via // Agent SDK. Gate-tier (deterministic stub server, fixed inputs); fires @@ -864,11 +841,6 @@ export const E2E_TOUCHFILES: Record = { 'plan-tune-inspect': ['test/session-runner-stream-lifecycle.test.ts', 'plan-tune/**', 'scripts/question-registry.ts', 'scripts/psychographic-signals.ts', 'scripts/one-way-doors.ts', 'bin/gstack-question-log', 'bin/gstack-question-preference', 'bin/gstack-developer-profile', 'test/skill-e2e-plan-tune.test.ts'], // /plan-tune cathedral (T16 — 5 E2E scenarios, all gate per D12) - 'plan-tune-hook-capture': ['hosts/claude/hooks/**', 'bin/gstack-question-log', 'bin/gstack-developer-profile', 'plan-tune/**', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'], - 'plan-tune-enforcement': ['hosts/claude/hooks/**', 'bin/gstack-question-preference', 'scripts/question-registry.ts', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'], - 'plan-tune-annotation': ['hosts/claude/hooks/**', 'scripts/declared-annotation.ts', 'scripts/psychographic-signals.ts', 'scripts/question-registry.ts', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'], - 'plan-tune-codex-import': ['bin/gstack-codex-session-import', 'bin/gstack-question-log', 'docs/spikes/codex-session-format.md', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'], - 'plan-tune-dream-cycle': ['bin/gstack-distill-free-text', 'bin/gstack-distill-apply', 'hosts/claude/hooks/**', 'plan-tune/**', 'test/skill-e2e-plan-tune-cathedral.test.ts', 'lib/jsonl-store.ts', 'lib/is-conductor.ts', 'test/plan-tune-cathedral-fixture.test.ts'], // Codex offering verification 'codex-offered-office-hours': ['test/session-runner-stream-lifecycle.test.ts', 'test/paid-retry-supervision.test.ts', 'office-hours/**', 'scripts/gen-skill-docs.ts', 'test/skill-e2e-plan.test.ts', @@ -1013,7 +985,6 @@ export const E2E_TOUCHFILES: Record = { ], // Gemini E2E — smoke test only (Gemini gets lost in worktrees on complex tasks) - 'gemini-smoke': ['scripts/gen-skill-docs.ts', 'test/helpers/gemini-session-runner.ts', 'lib/worktree.ts', 'test/gemini-e2e.test.ts'], // Coverage audit (shared fixture) + triage + gates @@ -1264,15 +1235,16 @@ export const E2E_TOUCHFILES: Record = { "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", ], + 'journey-negatives': ['test/session-runner-stream-lifecycle.test.ts', + 'test/skill-fixture.test.ts', + "test/plan-scope-recovery-av.test.ts", + "test/fixtures/plan-scope-recovery-av.json", + "test/fixtures/design-scope-checkpoint-at.json",'test/eng-scope-entry-ap.test.ts', '*/SKILL.md.tmpl', 'SKILL.md.tmpl', 'scripts/gen-skill-docs.ts', 'test/skill-routing-e2e.test.ts', + "test/design-scope-entry-aq.test.ts", + + "test/review-entry-and-design-clarity-au.test.ts", "scripts/resolvers/preamble/generate-preamble-bash.ts", "scripts/resolvers/preamble/generate-completion-status.ts", + ], - // Opus 4.7 behavior evals — keys match testName: values in the test file. - // Routing sub-tests use template literal `routing-${c.name}` testNames, - // which the touchfile completeness scanner skips; they inherit selection - // from the file-level touchfile entry via GLOBAL_TOUCHFILES. - 'fanout-arm-overlay-on': - ['test/session-runner-stream-lifecycle.test.ts', 'model-overlays/claude.md', 'model-overlays/opus-4-7.md', 'scripts/models.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-opus-47.test.ts'], - 'fanout-arm-overlay-off': - ['test/session-runner-stream-lifecycle.test.ts', 'model-overlays/claude.md', 'model-overlays/opus-4-7.md', 'scripts/models.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-opus-47.test.ts'], // Overlay efficacy harness (SDK) — measures whether overlay nudges change // behavior under @anthropic-ai/claude-agent-sdk (closer to real Claude Code @@ -1280,22 +1252,10 @@ export const E2E_TOUCHFILES: Record = { // completeness scanner doesn't require them; these entries exist for // diff-based selection accuracy. - // /ios-qa — agent flow E2E. Daemon + stub StateServer + codegen - // exercised end-to-end. The no-device path is gate-tier; the with-device - // path requires GSTACK_HAS_IOS_DEVICE=1 and is periodic-tier. - 'ios-qa-e2e': ['ios-qa/**', 'ios-fix/**', 'ios-design-review/**', 'ios-clean/**', 'ios-sync/**', 'test/skill-e2e-ios.test.ts'], - // Swift-build invariant test — requires the Swift toolchain. Compiles the - // fixture SPM package + runs the XCTest suite that validates the real - // Swift StateServer implementation (loopback bind, boot token rotation, - // session lock). Periodic-tier — Swift build is heavier than TS unit tests. - 'ios-qa-swift-build': ['ios-qa/templates/**', 'test/fixtures/ios-qa/FixtureApp/**', 'test/skill-e2e-ios-swift-build.test.ts'], // Real-device path — only runs with GSTACK_HAS_IOS_DEVICE=1 + a paired // iPhone. Validates the CoreDevice agent + iOS SDK toolchain. Periodic-tier. 'ios-qa-device': ['ios-qa/templates/**', 'test/fixtures/ios-qa/FixtureApp/**', 'test/skill-e2e-ios-device.test.ts'], - // /spec end-to-end via PTY — exercises the full Phase 1→5 pipeline - // including --execute spawn. Periodic-tier — paid + non-deterministic. - 'spec-execute': ['spec/**', 'scripts/resolvers/redact-doc.ts', 'scripts/resolvers/outside-voice.ts', 'scripts/resolvers/constants.ts', 'lib/outside-review-result.ts', 'bin/gstack-redact', 'lib/redact-engine.ts', 'test/skill-e2e-spec-execute.test.ts'], // /office-hours brain-writeback path under fake gbrain CLI (v1.50.0.0 // T7). Drives /office-hours with a regenerated SKILL.md that has the @@ -1370,16 +1330,10 @@ export const E2E_TOUCHFILES: Record = { 'test/fixtures/eng-omitted-select-361c.json', 'test/skill-e2e-plan-decision-classification.test.ts', 'test/fixtures/plan-decision-classification.ts', 'test/plan-review-calibration.test.ts', 'test/helpers/plan-review-decisions.ts', 'test/plan-review-decisions.test.ts', 'test/helpers/plan-review-cases.ts', 'test/plan-review-cases.test.ts', 'test/helpers/llm-judge.ts', 'lib/eval-model.ts', 'test/helpers/e2e-helpers.ts', 'test/helpers/eval-store.ts', 'test/helpers/eval-budgets.ts', 'scripts/resolvers/preamble/generate-ask-user-format.ts', 'docs/askuserquestion-split.md', 'plan-ceo-review/SKILL.md.tmpl', 'plan-ceo-review/sections/review-sections.md.tmpl', 'scripts/resolvers/tasks-section.ts'], 'health-reporting': ['health/**', 'test/skill-e2e-health.test.ts', 'test/helpers/health-eval-fixture.ts'], - 'codex-plan-ceo-format-mode': ['test/paid-retry-supervision.test.ts', 'plan-ceo-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts'], - 'codex-plan-ceo-format-approach': ['test/paid-retry-supervision.test.ts', 'plan-ceo-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts'], - 'codex-plan-eng-format-coverage': ['test/paid-retry-supervision.test.ts', 'scripts/resolvers/learnings.ts', 'test/review-entry-and-design-clarity-au.test.ts', 'test/fixtures/plan-scope-recovery-av.json', 'test/plan-scope-recovery-av.test.ts', 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts', 'test/plan-review-cases.test.ts'], - 'codex-plan-eng-format-kind': ['test/paid-retry-supervision.test.ts', 'scripts/resolvers/learnings.ts', 'test/review-entry-and-design-clarity-au.test.ts', 'test/fixtures/plan-scope-recovery-av.json', 'test/plan-scope-recovery-av.test.ts', 'test/eng-scope-entry-ap.test.ts', 'plan-eng-review/**', 'scripts/gen-skill-docs.ts', 'scripts/resolvers/**', 'model-overlays/gpt.md', 'model-overlays/gpt-5.4.md', 'test/helpers/codex-session-runner.ts', 'test/helpers/codex-eval.ts', 'test/codex-e2e-plan-format.test.ts', 'test/plan-review-cases.test.ts'], 'overlay-harness-claude-dedicated-tools-vs-bash': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'], 'overlay-harness-opus-4-7-effort-match-trivial': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'], 'overlay-harness-opus-4-7-literal-interpretation': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'], 'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'], - 'overlay-harness-opus-4-7-effort-match-trivial-sonnet': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial-sonnet.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'], - 'overlay-harness-opus-4-7-literal-interpretation-sonnet': ['model-overlays/**', 'test/fixtures/overlay-nudges.ts', 'test/helpers/agent-sdk-runner.ts', 'test/agent-sdk-runner.test.ts', 'scripts/resolvers/model-overlay.ts', 'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation-sonnet.test.ts', 'test/helpers/overlay-measurement.ts', 'test/helpers/overlay-workspace.ts', 'test/helpers/overlay-attempt.ts', 'test/overlay-measurement.test.ts', 'test/helpers/overlay-case.ts', 'test/helpers/overlay-case-policy.ts', 'test/helpers/overlay-lifecycle.ts', 'test/overlay-lifecycle.test.ts', 'test/overlay-sdk-cancel-eof.test.ts', 'test/overlay-recording-order.test.ts', 'test/paid-overlay-scheduling.test.ts', 'test/fixtures/overlay-admission-child.ts'], }; /** @@ -1503,7 +1457,6 @@ export const E2E_TIERS: Record = { // v1.21+ auto-mode regression tests 'office-hours-auto-mode': 'gate', 'auto-decide-preserved': 'periodic', - 'conductor-prose': 'periodic', // Real-PTY E2E batch — tier classification: // gate: cheap, deterministic, run on every PR @@ -1511,7 +1464,6 @@ export const E2E_TIERS: Record = { 'auq-format-gate': 'gate', // ~$0.50/run, native SDK question capture, single skill probe 'plan-ceo-mode-routing': 'periodic', // ~$3/run, deep navigation through 8-12 prior AskUserQuestions 'plan-design-with-ui-scope': 'gate', // ~$0.80/run - 'ship-idempotency-pty': 'periodic', // ~$3/run, real /ship in plan mode 'tpa-present': 'gate', // consent/credential safety guardrail; deterministic shims + grep asserts 'tpa-absent-linux': 'gate', // consent/credential safety guardrail; deterministic shims + grep asserts 'tpa-broken': 'gate', // consent/credential safety guardrail; deterministic shims + grep asserts @@ -1540,7 +1492,6 @@ export const E2E_TIERS: Record = { // Privacy gate for gstack-brain-sync — periodic (non-deterministic LLM call, // costs ~$0.30-$0.50 per run, not needed on every commit) - 'brain-privacy-gate': 'periodic', // /setup-gbrain Path 4 (Remote MCP) — periodic-tier. The stub HTTP // server is deterministic but the model's interpretation of "follow @@ -1578,11 +1529,6 @@ export const E2E_TIERS: Record = { 'plan-tune-inspect': 'gate', // /plan-tune cathedral (T16 per D12 — all gate) - 'plan-tune-hook-capture': 'gate', - 'plan-tune-enforcement': 'gate', - 'plan-tune-annotation': 'gate', - 'plan-tune-codex-import': 'gate', - 'plan-tune-dream-cycle': 'gate', // Codex offering verification 'codex-offered-office-hours': 'gate', @@ -1646,7 +1592,6 @@ export const E2E_TIERS: Record = { 'outside-voice-claude-code-to-codex': 'periodic', 'outside-plan-disabled-no-fallback': 'periodic', 'codex-sol-scope-termination': 'periodic', - 'gemini-smoke': 'periodic', // Design — gate for cheap functional, periodic for Opus/quality 'design-consultation-core': 'periodic', @@ -1706,10 +1651,9 @@ export const E2E_TIERS: Record = { 'journey-retro': 'periodic', 'journey-design-system': 'periodic', 'journey-visual-qa': 'periodic', + 'journey-negatives': 'periodic', // Opus 4.7 overlay evals — periodic (non-deterministic LLM behavior + Opus cost) - 'fanout-arm-overlay-on': 'periodic', - 'fanout-arm-overlay-off': 'periodic', // Overlay efficacy harness (SDK, paid) — periodic only @@ -1720,13 +1664,10 @@ export const E2E_TIERS: Record = { // on every Linux PR. Periodic keeps it in the weekly census on capable // hosts; re-promote if a macOS runner lands (flagged decision in the // test-infra overhaul plan). - 'ios-qa-e2e': 'periodic', // Swift toolchain only, no device required, but heavier than TS unit tests. - 'ios-qa-swift-build': 'periodic', // Requires a real connected + paired iPhone. Manual-trigger only. 'ios-qa-device': 'periodic', // /spec end-to-end PTY pipeline (paid, non-deterministic — periodic-tier). - 'spec-execute': 'periodic', // WS2 arm benchmark — periodic: full build-shaped agentic workflows, paid, // non-deterministic by construction (research instrument, not a gate). @@ -1737,16 +1678,10 @@ export const E2E_TIERS: Record = { 'plan-decision-classification': 'periodic', 'plan-devex-peer-comparison-classification': 'periodic', 'health-reporting': 'periodic', - 'codex-plan-ceo-format-mode': 'periodic', - 'codex-plan-ceo-format-approach': 'periodic', - 'codex-plan-eng-format-coverage': 'periodic', - 'codex-plan-eng-format-kind': 'periodic', 'overlay-harness-claude-dedicated-tools-vs-bash': 'periodic', 'overlay-harness-opus-4-7-effort-match-trivial': 'periodic', 'overlay-harness-opus-4-7-literal-interpretation': 'periodic', 'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'periodic', - 'overlay-harness-opus-4-7-effort-match-trivial-sonnet': 'periodic', - 'overlay-harness-opus-4-7-literal-interpretation-sonnet': 'periodic', }; /** @@ -1754,16 +1689,12 @@ export const E2E_TIERS: Record = { */ export const LLM_JUDGE_TOUCHFILES: Record = { 'setup-browser-cookies/SKILL.md workflow': ['setup-browser-cookies/SKILL.md.tmpl', 'setup-browser-cookies/SKILL.md', 'BROWSER.md', 'test/helpers/cookie-workflow-judge-input.ts', 'test/cookie-workflow-judge-input.test.ts', 'test/helpers/cookie-workflow-manual-review.ts', 'test/cookie-workflow-manual-review.test.ts', 'test/helpers/manual-judge-review-fixture.ts', '.github/cookie-workflow-manual-review.json', 'test/helpers/workflow-judge-input.ts', 'test/skill-llm-eval.test.ts'], - 'command reference table': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/commands.ts', 'gstack/llms.txt', 'test/skill-llm-eval.test.ts'], - 'snapshot flags reference': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/snapshot.ts', 'test/skill-llm-eval.test.ts'], - 'browse/SKILL.md reference': ['browse/sections/**', 'browse/SKILL.md', 'browse/SKILL.md.tmpl', 'browse/src/**', 'test/skill-llm-eval.test.ts'], + 'browse/SKILL.md reference': ['browse/sections/**', 'browse/SKILL.md', 'browse/SKILL.md.tmpl', 'browse/src/**', 'SKILL.md', 'SKILL.md.tmpl', 'gstack/llms.txt', 'test/fixtures/eval-baselines.json', 'test/skill-llm-eval.test.ts'], 'setup block': ['browse/SKILL.md', 'browse/SKILL.md.tmpl', 'scripts/resolvers/aside.ts', 'scripts/resolvers/browse.ts', 'test/skill-llm-eval.test.ts'], - 'regression vs baseline': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'browse/src/commands.ts', 'test/fixtures/eval-baselines.json', 'test/skill-llm-eval.test.ts'], 'qa/SKILL.md workflow': ['qa/sections/**', 'qa/SKILL.md', 'qa/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'], 'qa/SKILL.md health rubric': ['qa/sections/**', 'qa/SKILL.md', 'qa/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'], 'qa/SKILL.md anti-refusal': ['qa/sections/**', 'qa/SKILL.md', 'qa/SKILL.md.tmpl', 'qa-only/SKILL.md', 'qa-only/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'], 'cross-skill greptile consistency': ['review/SKILL.md', 'review/SKILL.md.tmpl', 'ship/SKILL.md', 'ship/SKILL.md.tmpl', 'review/greptile-triage.md', 'retro/SKILL.md', 'retro/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts'], - 'baseline score pinning': ['browse/sections/**', 'SKILL.md', 'SKILL.md.tmpl', 'test/fixtures/eval-baselines.json', 'test/skill-llm-eval.test.ts'], // Ship & Release 'ship/SKILL.md workflow': ['ship/SKILL.md', 'ship/SKILL.md.tmpl', 'test/skill-llm-eval.test.ts', 'test/helpers/workflow-judge-input.ts', 'test/helpers/workflow-judge-cache.ts', 'test/workflow-judge-cache.test.ts', 'scripts/eval-input-cache.ts', 'test/eval-input-cache.test.ts', 'test/workflow-judge-input.test.ts', 'test/helpers/workflow-excerpt.ts', diff --git a/test/hermetic-wiring.test.ts b/test/hermetic-wiring.test.ts index 78fa96679..a3efeda2a 100644 --- a/test/hermetic-wiring.test.ts +++ b/test/hermetic-wiring.test.ts @@ -25,7 +25,6 @@ const RUNNERS = [ 'test/helpers/session-runner.ts', 'test/helpers/claude-pty-runner.ts', 'test/helpers/codex-session-runner.ts', - 'test/helpers/gemini-session-runner.ts', 'test/helpers/agent-sdk-runner.ts', ]; diff --git a/test/skill-e2e-ios-swift-build.test.ts b/test/ios-qa-swift-build.test.ts similarity index 97% rename from test/skill-e2e-ios-swift-build.test.ts rename to test/ios-qa-swift-build.test.ts index 529c6348a..fca45c211 100644 --- a/test/skill-e2e-ios-swift-build.test.ts +++ b/test/ios-qa-swift-build.test.ts @@ -1,6 +1,7 @@ // Swift-build invariant tests. Runs against the fixture iOS app at // test/fixtures/ios-qa/FixtureApp/. Requires the Swift toolchain -// (Xcode CLI tools or stand-alone Swift). Skipped if swift is not on PATH. +// (Xcode CLI tools or stand-alone Swift). The swift build invariants run +// only with GSTACK_TEST_SWIFT=1; the parity and harness pins always run. // // Two invariants: // @@ -298,13 +299,9 @@ describe('iOS tap harness regressions', () => { }); }); -function hasSwift(): boolean { - const r = spawnSync('swift', ['--version'], { stdio: 'pipe', timeout: 30_000 }); - return r.status === 0; -} - -const swiftAvailable = hasSwift(); -const describeIfSwift = swiftAvailable ? describe : describe.skip; +// Explicit opt-in, not tool presence: free shards run on hosts where Swift +// may be installed, and a full swift build does not belong in every PR run. +const describeIfSwift = process.env.GSTACK_TEST_SWIFT === '1' ? describe : describe.skip; describeIfSwift('swift build invariants', () => { // DebugBridgeUI + DebugBridgeTouch are iOS-only (they link UIKit). Plain diff --git a/test/skill-e2e-ios.test.ts b/test/ios-qa.test.ts similarity index 95% rename from test/skill-e2e-ios.test.ts rename to test/ios-qa.test.ts index 56211637a..194cc13fd 100644 --- a/test/skill-e2e-ios.test.ts +++ b/test/ios-qa.test.ts @@ -1,12 +1,9 @@ // High-level E2E for /ios-qa skill flow. // -// Two scenarios: -// 1. NO_DEVICE (gate-tier compatible): runs the gen-accessors codegen -// against a SwiftUI fixture, verifies output is correct, no daemon -// hardware required. Catches regression in source-read + codegen + -// cache + render paths without an iPhone. -// 2. WITH_DEVICE (periodic-tier, requires GSTACK_HAS_IOS_DEVICE=1): full -// daemon + tailnet + USB tunnel loop. Skipped in CI. +// Runs the gen-accessors codegen against a SwiftUI fixture and simulates the +// agent flow against the daemon with a fake device tunnel — no hardware. +// Catches regression in source-read + codegen + cache + render paths without +// an iPhone. The real-device loop lives in test/skill-e2e-ios-device.test.ts. // // Note: The detailed daemon HTTP unit/integration tests live next to the // daemon source (ios-qa/daemon/test/*). This file tests the agent-flow @@ -22,7 +19,6 @@ import type { DeviceTunnel } from '../ios-qa/daemon/src/proxy'; import { grantIdentity } from '../ios-qa/daemon/src/allowlist'; import { generate } from '../ios-qa/scripts/gen-accessors'; -const HAS_DEVICE = process.env.GSTACK_HAS_IOS_DEVICE === '1'; const DEVICE_TOKEN = 'rotated-mock-bearer-token'; @@ -494,14 +490,3 @@ describe('ios-qa E2E (agent-flow simulation)', () => { } }); }); - -// ───────── WITH_DEVICE — manual smoke tests (skipped in CI) ───────── - -(HAS_DEVICE ? describe : describe.skip)('ios-qa E2E (with device)', () => { - test('WITH_DEVICE: full agent loop against a real iPhone', () => { - const workDir = makeWorkDir(); - // Stub — real implementation requires `devicectl` + an attached iPhone. - // Documented in ios-qa/SKILL.md.tmpl under "Manual smoke test". - expect(HAS_DEVICE).toBe(true); - }); -}); diff --git a/test/skill-e2e-memory-pipeline.test.ts b/test/memory-pipeline.test.ts similarity index 100% rename from test/skill-e2e-memory-pipeline.test.ts rename to test/memory-pipeline.test.ts diff --git a/test/native-auto-decide.test.ts b/test/native-auto-decide.test.ts index 367d5a4aa..a99e1df2d 100644 --- a/test/native-auto-decide.test.ts +++ b/test/native-auto-decide.test.ts @@ -98,7 +98,7 @@ test('the native reader cannot promote foreign cwd, child, user or tool-result t }); test('new native annotation dependencies retain all existing observation caller owners',()=>{ - const expected=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','office-hours-auto-mode','auto-decide-preserved','conductor-prose']; + const expected=['plan-ceo-review-plan-mode','plan-eng-review-plan-mode','plan-design-review-plan-mode','plan-devex-review-plan-mode','plan-mode-no-op','office-hours-auto-mode','auto-decide-preserved']; for(const file of ['test/helpers/native-auto-decide.ts','test/native-auto-decide.test.ts','test/native-auto-decide-pty.test.ts','test/fixtures/native-auto-decide-ag.json']){ const owners=Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes(file)).map(([name])=>name);expect(owners).toEqual(expected); expect(selectTests([file],E2E_TOUCHFILES).selected).toContain('auto-decide-preserved'); diff --git a/test/overlay-lifecycle.test.ts b/test/overlay-lifecycle.test.ts index a81c0ef10..a14d76dac 100644 --- a/test/overlay-lifecycle.test.ts +++ b/test/overlay-lifecycle.test.ts @@ -27,8 +27,8 @@ test('every fixture owns one paid wrapper; public models/trials/concurrency/turn } expect(isPaidTestFile('test/overlay-lifecycle.test.ts')).toBe(false); expect(OVERLAY_FIXTURES.every(f => f.trials === 10 && f.concurrency === 3)).toBe(true); - expect(OVERLAY_FIXTURES.map(f => f.maxTurns ?? 5)).toEqual([15,8,15,15,8,15]); - expect(OVERLAY_FIXTURES.map(f => f.model)).toEqual([...Array(3).fill('claude-opus-4-7'), ...Array(3).fill('claude-sonnet-4-6')]); + expect(OVERLAY_FIXTURES.map(f => f.maxTurns ?? 5)).toEqual([15,8,15,15]); + expect(OVERLAY_FIXTURES.map(f => f.model)).toEqual([...Array(3).fill('claude-opus-4-7'), 'claude-sonnet-4-6']); expect([OVERLAY_CASE_WORK_MS, OVERLAY_RECORD_GRACE_MS, OVERLAY_CASE_OUTER_MS, OVERLAY_MIN_FILE_WALL_MS]).toEqual([1_800_000, 5_000, 1_810_000, 1_830_000]); }); diff --git a/test/overlay-measurement.test.ts b/test/overlay-measurement.test.ts index 2e22a204c..7ecfedd8b 100644 --- a/test/overlay-measurement.test.ts +++ b/test/overlay-measurement.test.ts @@ -226,11 +226,11 @@ describe('literal fixture task correctness', () => { expect(() => assertReadOnlyWorkspace(before, snapshotWorkspace(dir))).toThrow('b'); })); test('registry retires absent-nudge fanout cases and preserves remaining model/trial budgets', () => { - expect(OVERLAY_FIXTURES).toHaveLength(6); + expect(OVERLAY_FIXTURES).toHaveLength(4); expect(OVERLAY_FIXTURES.some((f) => f.id.includes('fanout') || f.comparison?.unsupportedHypothesis)).toBe(false); expect(OVERLAY_FIXTURES.every((f) => f.trials === 10 && f.comparison)).toBe(true); expect(OVERLAY_FIXTURES.filter((f) => f.model === 'claude-opus-4-7')).toHaveLength(3); - expect(OVERLAY_FIXTURES.filter((f) => f.model === 'claude-sonnet-4-6')).toHaveLength(3); + expect(OVERLAY_FIXTURES.filter((f) => f.model === 'claude-sonnet-4-6')).toHaveLength(1); }); }); diff --git a/test/paid-overlay-scheduling.test.ts b/test/paid-overlay-scheduling.test.ts index 5a126295e..8a8f43297 100644 --- a/test/paid-overlay-scheduling.test.ts +++ b/test/paid-overlay-scheduling.test.ts @@ -52,7 +52,7 @@ describe('overlay file policy', () => { }); test('only the exact wrapper family gets one attempt and the extra process grace', () => { - expect(overlayFiles).toHaveLength(6); + expect(overlayFiles).toHaveLength(4); expect(OVERLAY_MAX_ACTIVE_SHARDS).toBe(1); expect(OVERLAY_MIN_FILE_WALL_MS).toBe(1_830_000); for (const file of overlayFiles) { @@ -123,16 +123,16 @@ describe('overlay file policy', () => { }); describe('overlay manifest affinity and CI capacity', () => { - test('actual lifecycle wrapper guards include all six in periodic and exclude all six from gate', () => { + test('actual lifecycle wrapper guards include all four in periodic and exclude all four from gate', () => { for (const tier of ['periodic', 'gate'] as const) { const manifest = buildRunManifest({ tier, sliceCount: 6, evalsAll: true, env: { EVALS_ALL: '1' } }); const entries = manifest.entries.filter(entry => isOverlayTestFile(entry.file)); - expect(entries).toHaveLength(6); + expect(entries).toHaveLength(4); expect(entries.every(entry => entry.status === (tier === 'periodic' ? 'planned' : 'excluded'))).toBe(true); } }); - test('95 files retain every case, reserve slice six, and fit 330 minutes with actual family walls', () => { + test('93 files retain every case, reserve slice six, and fit 330 minutes with actual family walls', () => { const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'overlay-affinity-')); const normalFiles = Array.from({ length: 89 }, (_, i) => `test/skill-e2e-normal-${i.toString().padStart(2, '0')}.test.ts`); const discovered = [...normalFiles, ...overlayFiles]; @@ -144,11 +144,11 @@ describe('overlay manifest affinity and CI capacity', () => { } const opts = { tier: 'periodic' as const, sliceCount: 6, evalsAll: true, discovered, rootDir: dir, env: { EVALS_ALL: '1' } }; const manifest = buildRunManifest(opts); - expect(manifest.entries).toHaveLength(95); - expect(new Set(manifest.entries.map(e => e.file)).size).toBe(95); + expect(manifest.entries).toHaveLength(93); + expect(new Set(manifest.entries.map(e => e.file)).size).toBe(93); expect(manifest.entries.every(e => e.status === 'planned')).toBe(true); const counts = [1, 2, 3, 4, 5, 6].map(slice => manifest.entries.filter(e => e.slice === slice).length); - expect(counts).toEqual([18, 18, 18, 18, 17, 6]); + expect(counts).toEqual([18, 18, 18, 18, 17, 4]); expect(manifest.entries.filter(e => e.slice === 6).map(e => e.file).sort()).toEqual([...overlayFiles].sort()); expect(buildRunManifest({ ...opts, discovered: [...discovered].reverse() })).toEqual(manifest); expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest); @@ -172,7 +172,7 @@ describe('overlay manifest affinity and CI capacity', () => { const overlayMinutes = Math.ceil(overlayFiles.length / OVERLAY_MAX_ACTIVE_SHARDS) * Math.max(...overlayFiles.map(file => resolvePaidShardTimeoutMs([file]))) / 60_000; expect(normalMinutes).toBe(270); - expect(overlayMinutes).toBe(183); + expect(overlayMinutes).toBe(122); expect(job['timeout-minutes']).toBeGreaterThanOrEqual(Math.max(normalMinutes, overlayMinutes) + 20); expect(job['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(AUTOPLAN_CHAIN_BUDGET.ciJobMs); diff --git a/test/paid-pr-profile.test.ts b/test/paid-pr-profile.test.ts index 5eb6ff638..e1c5eca2a 100644 --- a/test/paid-pr-profile.test.ts +++ b/test/paid-pr-profile.test.ts @@ -18,7 +18,7 @@ import { resolveModuleSelection } from './helpers/e2e-helpers'; const ROOT = path.resolve(import.meta.dir, '..'); const CEO_FILES = [ 'test/skill-e2e-plan.test.ts', 'test/skill-e2e-ask-user-question-format-compliance.test.ts', - 'test/skill-e2e-opus-47.test.ts', 'test/skill-llm-eval.test.ts', + 'test/skill-e2e-retro.test.ts', 'test/skill-llm-eval.test.ts', ]; const ceoManifest = () => buildRunManifest({ tier: 'gate', profile: 'pr', sliceCount: 1, evalsAll: false, env: {}, changedFiles: ['plan-ceo-review/SKILL.md.tmpl'], discovered: CEO_FILES }); @@ -70,7 +70,7 @@ describe('PR profile paid-runner integration', () => { expect(manifest.selection?.e2e).toContain('plan-ceo-review-benefits'); expect(manifest.selection?.e2e).not.toContain('plan-ceo-review-plan-mode'); expect(manifest.selection?.judges).toContain('plan-ceo-review/SKILL.md modes'); - expect(manifest.entries.find(entry => entry.file.includes('opus-47'))?.status).toBe('skipped-by-diff'); + expect(manifest.entries.find(entry => entry.file.includes('skill-e2e-retro'))?.status).toBe('skipped-by-diff'); expect(manifest.entries.find(entry => entry.file.includes('ask-user-question'))?.status).toBe('planned'); expect(manifest.prCoverage?.deferred.some(item => item.id === 'plan-ceo-review-plan-mode')).toBe(true); expect(parseRunManifest(JSON.stringify(manifest))).toEqual(manifest); @@ -145,8 +145,8 @@ describe('PR profile paid-runner integration', () => { broad.selection!.e2e!.push('qa-only-no-fix'); broad.prCoverage!.e2e.push('qa-only-no-fix'); expect(() => parseRunManifest(JSON.stringify(broad))).toThrow('broad-only'); const injected = structuredClone(manifest); - injected.entries.find(entry => entry.file.includes('opus-47'))!.status = 'planned'; - injected.entries.find(entry => entry.file.includes('opus-47'))!.slice = 1; + injected.entries.find(entry => entry.file.includes('skill-e2e-retro'))!.status = 'planned'; + injected.entries.find(entry => entry.file.includes('skill-e2e-retro'))!.slice = 1; expect(() => parseRunManifest(JSON.stringify(injected))).toThrow('outside its PR case selection'); for (const action of ['remove', 'skip', 'duplicate'] as const) { const missing = structuredClone(manifest); @@ -184,7 +184,7 @@ describe('PR profile paid-runner integration', () => { const judge = new RegExp(prProfileTestNamePattern('test/skill-llm-eval.test.ts', selection)); expect(judge.test('LLM-as-judge plan-ceo-review/SKILL.md modes')).toBe(true); expect(judge.test('LLM-as-judge plan-ceo-review/SKILLxmd modes')).toBe(false); - expect(prProfileFileSelected('test/skill-e2e-opus-47.test.ts', selection)).toBe(false); + expect(prProfileFileSelected('test/skill-e2e-retro.test.ts', selection)).toBe(false); }); test('real Bun child executes only the persisted case through the actual registered helper', async () => { diff --git a/test/paid-retry-supervision.test.ts b/test/paid-retry-supervision.test.ts index ec37e7df5..be16534ac 100644 --- a/test/paid-retry-supervision.test.ts +++ b/test/paid-retry-supervision.test.ts @@ -13,9 +13,8 @@ import { const read = (file: string) => readFileSync(join(import.meta.dir, '..', file), 'utf8'); const newBudgets = FILE_RETRY_BUDGETS.filter(row => !FINDING_RETRY_BUDGETS.some(old => old.file === row.file)); const expectedWalls = { - 'test/skill-llm-eval.test.ts': 6_920_000, + 'test/skill-llm-eval.test.ts': 5_960_000, 'test/skill-e2e-auq-consistency.test.ts': 2_040_000, - 'test/codex-e2e-plan-format.test.ts': 5_000_000, 'test/skill-e2e-auq-matrix.test.ts': 3_720_000, 'test/skill-e2e-plan-format.test.ts': 2_600_000, 'test/skill-e2e-auto-decide-preserved.test.ts': 1_920_000, @@ -30,9 +29,9 @@ const expectedWalls = { 'test/skill-e2e-plan.test.ts': 7_320_000, }; -test('registration covers exactly the fifteen demonstrated full-file retry gaps', () => { +test('registration covers exactly the fourteen demonstrated full-file retry gaps', () => { expect(Object.fromEntries(newBudgets.map(row => [row.file, row.shardMs]))).toEqual(expectedWalls); - expect(new Set(FILE_RETRY_BUDGETS.map(row => row.file)).size).toBe(21); + expect(new Set(FILE_RETRY_BUDGETS.map(row => row.file)).size).toBe(20); expect(STRICT_RETRY_CASE_BUDGETS.map(row => row.file)).toEqual([ ...FINDING_RETRY_BUDGETS.map(row => row.file), AUQ_CONSISTENCY_RETRY_BUDGET.file, ]); @@ -51,8 +50,6 @@ test('source allowances retain all captures, cases, and finalization grace', () expect(auq).toContain('for (const [i, capture] of captures.entries())'); expect(auq).toContain('N_RUNS * CAPTURE_MS + 60_000'); expect(AUQ_CONSISTENCY_RETRY_BUDGET.testMs).toBe(960_000); - expect(timeoutExpressions('test/codex-e2e-plan-format.test.ts')).toEqual(Array(4).fill('CAPTURE_LONG_MS')); - expect(read('test/codex-e2e-plan-format.test.ts')).toContain('test(name, fn, timeout + CODEX_EVAL_FINALIZE_MS)'); expect(read('test/helpers/codex-eval.ts')).toContain('CODEX_EVAL_FINALIZE_MS = 2 * CODEX_DRAIN_GRACE_MS'); expect(read('test/helpers/codex-session-runner.ts')).toMatch(/CODEX_DRAIN_GRACE_MS\s*=\s*5_000/); expect(read('test/helpers/office-hours-attempt.ts')).toContain('OFFICE_HOURS_BUN_GRACE_MS = 10_000'); @@ -140,15 +137,15 @@ test('quality judge supervision includes the added judge without changing ordina expect(ALL_TIERS).toEqual({ JUDGE_MS: 120000, CAPTURE_MS: 300000, CAPTURE_LONG_MS: 600000, PTY_MS: 900000, PTY_LONG_MS: 1200000 }); const quality = 'test/skill-llm-eval.test.ts'; const qualityBudget = FILE_RETRY_BUDGETS.find(row => row.file === quality)!; - expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 6_920_000, source: 'registered', policyId: qualityBudget.id }); + expect(resolvePaidShardBudget([quality])).toEqual({ timeoutMs: 5_960_000, source: 'registered', policyId: qualityBudget.id }); expect(retriesForFiles([quality])).toBe(1); const qualitySource = read(quality); const judgeTimeouts = [...qualitySource.matchAll(/}\s*,\s*(JUDGE_MS|WORKFLOW_JUDGE_TEST_MS)\s*\);/g)].map(match => match[1]); - expect(judgeTimeouts.filter(timeout => timeout === 'JUDGE_MS')).toHaveLength(11); + expect(judgeTimeouts.filter(timeout => timeout === 'JUDGE_MS')).toHaveLength(7); expect(judgeTimeouts.filter(timeout => timeout === 'WORKFLOW_JUDGE_TEST_MS')).toHaveLength(16); expect(qualitySource).toContain('WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000'); expect(qualitySource).toContain('const workDeadline = started + JUDGE_MS'); - expect(qualityBudget.shardMs).toBe((11 * ALL_TIERS.JUDGE_MS + 16 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000); + expect(qualityBudget.shardMs).toBe((7 * ALL_TIERS.JUDGE_MS + 16 * (ALL_TIERS.JUDGE_MS + 10_000)) * 2 + 120_000); expect(FINDING_RETRY_BUDGETS.map(row => [row.cases, row.testMs, row.retries, row.shardMs])).toEqual([ [2, 1500000, 1, 6120000], ...Array(5).fill([1, 1500000, 1, 3120000]), ]); @@ -213,9 +210,9 @@ test('both gate executors cover the complete census without increasing aggregate expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: slices }, (_, i) => i + 1)); expect(planned.slices).toBe(slices); const manifest = buildRunManifest({ tier: 'gate', sliceCount: planned.slices, evalsAll: true, env: { EVALS_ALL: '1' } }); - expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(58); + expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(52); const files = manifest.entries.filter(row => row.status === 'planned').map(row => row.file); - expect(new Set(files).size).toBe(58); + expect(new Set(files).size).toBe(52); expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected.sort()); const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs( manifest.entries.filter(row => row.status === 'planned' && row.slice === slice).map(row => row.file), workers, @@ -250,8 +247,8 @@ test('the periodic executor supervises every actual case and retry within its CI const manifest = buildRunManifest({ tier: 'periodic', sliceCount: planned.slices, dedicatedAutoplanSlice: planned.dedicatedAutoplanSlice, evalsAll: true, env: { EVALS_ALL: '1' } }); const census = manifest.entries.filter(row => row.status === 'planned'); - expect(census).toHaveLength(100); - expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_920_000); + expect(census).toHaveLength(82); + expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(5_960_000); expect(manifest.autoplanSlice).toBe(8); const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs( census.filter(row => row.slice === slice).map(row => row.file), active.jobs, diff --git a/test/paid-shards.test.ts b/test/paid-shards.test.ts index a73922deb..b4577d0c4 100644 --- a/test/paid-shards.test.ts +++ b/test/paid-shards.test.ts @@ -60,7 +60,7 @@ describe('paid test enumeration', () => { const files = collectPaidTestFiles(); expect(files.length).toBeGreaterThan(0); expect(files.every(isPaidTestFile)).toBe(true); - expect(PAID_TEST_GLOBS.length).toBe(7); + expect(PAID_TEST_GLOBS.length).toBe(6); const shards = planPaidShards(files); expect(shards.flat().sort()).toEqual([...files].sort()); diff --git a/test/periodic-fixture-selection.test.ts b/test/periodic-fixture-selection.test.ts index 810490ece..546fc5815 100644 --- a/test/periodic-fixture-selection.test.ts +++ b/test/periodic-fixture-selection.test.ts @@ -93,7 +93,7 @@ describe('periodic fixture dependencies select their behavioral cases', () => { ['test/fixtures/ceo-current-record-6aef.json', ['plan-ceo-finding-count']], ['test/fixtures/ceo-payment-ledger-decisions.json', ['plan-ceo-finding-count']], ['test/setup-gbrain-remote-caller.test.ts', ['setup-gbrain-remote']], - ['test/skill-fixture.test.ts', ['journey-ideation', 'journey-plan-eng', 'journey-debug', 'journey-qa', 'journey-code-review', 'journey-ship', 'journey-docs', 'journey-retro', 'journey-design-system', 'journey-visual-qa']], + ['test/skill-fixture.test.ts', ['journey-ideation', 'journey-plan-eng', 'journey-debug', 'journey-qa', 'journey-code-review', 'journey-ship', 'journey-docs', 'journey-retro', 'journey-design-system', 'journey-visual-qa', 'journey-negatives']], ['test/office-hours-writeback-env.test.ts', ['office-hours-brain-writeback']], ['test/helpers/setup-gbrain-sandbox.ts', ['setup-gbrain-bad-token', 'setup-gbrain-path4-local-pglite', 'setup-gbrain-remote']], ['test/helpers/setup-gbrain-fixture-command.ts', ['setup-gbrain-bad-token', 'setup-gbrain-path4-local-pglite']], @@ -154,7 +154,7 @@ describe('periodic fixture dependencies select their behavioral cases', () => { } test('SDK runner changes retain the native gate and existing periodic consumers', () => { - const periodic = ['brain-privacy-gate', 'setup-gbrain-remote', 'setup-gbrain-bad-token', + const periodic = ['setup-gbrain-remote', 'setup-gbrain-bad-token', 'setup-gbrain-path4-local-pglite', ...OVERLAY_FIXTURES.map(fixture => `overlay-harness-${fixture.id}`)]; const result = selectTests(['test/agent-sdk-runner.test.ts'], E2E_TOUCHFILES); expect(result.reason).toBe('diff'); @@ -251,9 +251,9 @@ test('native fixture dependencies include the migrated auto-decision and seeded test('shared native input dependencies select every PTY consumer without changing tiers', () => { const expected = selectTests(['test/helpers/claude-pty-runner.ts'], E2E_TOUCHFILES).selected.sort(); - expect(expected).toHaveLength(22); + expect(expected).toHaveLength(20); expect(expected.filter(id => E2E_TIERS[id] === 'gate')).toHaveLength(7); - expect(expected.filter(id => E2E_TIERS[id] === 'periodic')).toHaveLength(15); + expect(expected.filter(id => E2E_TIERS[id] === 'periodic')).toHaveLength(13); for (const file of ['test/plan-count-design-ui-recovery.test.ts', 'test/fixtures/design-ui-boxed-question.json', 'test/pty-workspace-trust.test.ts', 'test/fixtures/pty-companion-cli.ts', 'test/helpers/plan-skill-questions.ts', 'test/plan-skill-questions.test.ts', 'test/fixtures/design-tasks-bash-permission.json', 'test/fixtures/eng-auq-validation-error.json', 'test/helpers/plan-skill-question-events.ts', 'test/plan-skill-question-events.test.ts', @@ -266,14 +266,14 @@ test('shared native input dependencies select every PTY consumer without changin test('seed submission dependencies select every seeded caller with its existing tier', () => { - const expected = ['auto-decide-preserved', 'conductor-prose', 'plan-ceo-review-plan-mode', + const expected = ['auto-decide-preserved', 'plan-ceo-review-plan-mode', 'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-mode-no-op']; for (const file of ['test/helpers/fake-plan-seed.ts', 'test/helpers/plan-seed-submission.ts', 'test/plan-seed-submission.test.ts', 'test/fixtures/plan-seed-cli.ts']) { const result = selectTests([file], E2E_TOUCHFILES); expect(result.reason).toBe('diff'); expect(result.selected.sort()).toEqual(expected); } - expect(expected.map(id => E2E_TIERS[id])).toEqual(['periodic', 'periodic', 'gate', 'periodic', 'gate', 'periodic', 'gate']); + expect(expected.map(id => E2E_TIERS[id])).toEqual(['periodic', 'gate', 'periodic', 'gate', 'periodic', 'gate']); }); test('task emission source selects CEO completion consumers', () => { @@ -323,7 +323,6 @@ test('Eng approval-rule source and free contract controls select every declared 'plan-review-report', 'plan-eng-review-plan-mode', 'plan-mode-no-op', - 'conductor-prose', 'carve-section-loading', 'autoplan-chain-pty', 'plan-eng-finding-count', @@ -334,8 +333,6 @@ test('Eng approval-rule source and free contract controls select every declared 'plan-ceo-review-prosons-cadence', 'plan-review-prosons-format', 'codex-offered-eng-review', - 'codex-plan-eng-format-coverage', - 'codex-plan-eng-format-kind', 'plan-eng-coverage-audit', 'autoplan-dual-voice' ]; @@ -363,8 +360,7 @@ test('native compact-boundary ancestry selects every consuming callback', () => 'plan-eng-finding-count', 'plan-design-finding-count', 'plan-devex-finding-count', 'plan-eng-multi-finding-batching', 'plan-ceo-split-overflow', 'plan-design-with-ui-scope', 'plan-design-review-plan-mode', 'plan-eng-review-plan-mode', - 'auto-decide-preserved', 'conductor-prose', - ].sort(); + 'auto-decide-preserved', ].sort(); for (const file of ['test/helpers/plan-count-transcript.ts', 'test/plan-count-session-cwd.test.ts']) { const selected = selectTests([file], E2E_TOUCHFILES); expect(selected.reason).toBe('diff'); @@ -383,7 +379,7 @@ test('same-plan expansion disposition replay selects the existing mode helper co test('structured auto-decision evidence selects every native observer', () => { - const expected = ['auto-decide-preserved', 'conductor-prose', 'plan-ceo-review-plan-mode', + const expected = ['auto-decide-preserved', 'plan-ceo-review-plan-mode', 'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-mode-no-op']; for (const file of ['test/auto-decide-structured.test.ts', 'test/fixtures/auto-decide-structured-77.json', 'test/helpers/auto-decision-state.ts', 'test/auto-decision-state.test.ts', 'test/fixtures/auto-decide-state-cab3.json']) { @@ -398,7 +394,7 @@ test('structured auto-decision evidence selects every native observer', () => { test('explanatory native mode evidence selects all observers with their existing tiers', () => { - const expected = ['auto-decide-preserved', 'conductor-prose', 'office-hours-auto-mode', + const expected = ['auto-decide-preserved', 'office-hours-auto-mode', 'plan-ceo-review-plan-mode', 'plan-design-review-plan-mode', 'plan-devex-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-mode-no-op']; for (const file of ['test/helpers/native-auto-decide.ts', 'test/auto-decide-current-declaration.test.ts', @@ -408,7 +404,7 @@ test('explanatory native mode evidence selects all observers with their existing expect(selectTests([file], LLM_JUDGE_TOUCHFILES).selected).toEqual([]); } expect(expected.map(id => E2E_TIERS[id])).toEqual([ - 'periodic', 'periodic', 'gate', 'gate', 'periodic', 'gate', 'periodic', 'gate', + 'periodic', 'gate', 'gate', 'periodic', 'gate', 'periodic', 'gate', ]); }); @@ -432,8 +428,7 @@ test('file supervision regression selects all affected callers with their existi 'plan-ceo-review-benefits', 'plan-devex-finding-floor', 'plan-mode-no-op', 'plan-review-report', ]; const periodic = [ - 'auto-decide-preserved', 'codex-plan-ceo-format-approach', 'codex-plan-ceo-format-mode', - 'codex-plan-eng-format-coverage', 'codex-plan-eng-format-kind', 'plan-ceo-mode-routing', + 'auto-decide-preserved', 'plan-ceo-mode-routing', 'plan-ceo-review', 'plan-ceo-review-expansion-energy', 'plan-ceo-review-format-approach', 'plan-ceo-review-format-mode', 'plan-ceo-review-prosons-cadence', 'plan-ceo-review-selective', 'plan-design-finding-floor', 'plan-eng-finding-floor', 'plan-eng-review', 'plan-eng-review-artifact', @@ -498,7 +493,6 @@ const nativeRepairDependencies = [ "plan-mode-no-op", "office-hours-auto-mode", "auto-decide-preserved", - "conductor-prose" ] }, { @@ -595,10 +589,8 @@ const nativeRepairDependencies = [ "plan-mode-no-op", "office-hours-auto-mode", "auto-decide-preserved", - "conductor-prose", "plan-ceo-mode-routing", "plan-design-with-ui-scope", - "ship-idempotency-pty", "autoplan-chain-pty", "plan-ceo-finding-count", "plan-eng-finding-count", @@ -632,7 +624,6 @@ test('native repair dependencies preserve every original tier', () => { "plan-mode-no-op": "gate", "office-hours-auto-mode": "gate", "auto-decide-preserved": "periodic", - "conductor-prose": "periodic", "plan-ceo-finding-count": "periodic", "plan-eng-finding-count": "periodic", "plan-design-finding-count": "periodic", @@ -657,7 +648,6 @@ test('promoted public transcript decoder keeps its actual callers selected', () 'plan-eng-review-plan-mode', 'plan-design-review-plan-mode', 'auto-decide-preserved', - 'conductor-prose', 'plan-ceo-mode-routing', 'plan-design-with-ui-scope', 'autoplan-chain-pty', @@ -749,7 +739,7 @@ test('numbered native-menu captures select the existing parser consumers', () => const expected = Object.entries(E2E_TOUCHFILES) .filter(([, files]) => files.includes('test/plan-skill-questions.test.ts')) .map(([id]) => id).sort(); - expect(expected).toHaveLength(22); + expect(expected).toHaveLength(20); for (const file of ['test/pty-numbered-option-indent-native.test.ts', 'test/fixtures/ceo-split-e5-numbered-description-491.json']) { expect(selectTests([file], E2E_TOUCHFILES).selected.sort()).toEqual(expected); @@ -808,7 +798,7 @@ test('stderr lifecycle regression selects runtime consumers without a quality-ma 'benchmark-workflow', 'setup-deploy-workflow', 'autoplan-dual-voice', 'scrape-match-path', 'scrape-prototype-path', 'skillify-happy-path', 'skillify-provenance-refusal', 'skillify-approval-reject', 'journey-ideation', 'journey-plan-eng', 'journey-debug', 'journey-qa', 'journey-code-review', 'journey-ship', 'journey-docs', - 'journey-retro', 'journey-design-system', 'journey-visual-qa', 'fanout-arm-overlay-on', 'fanout-arm-overlay-off', + 'journey-retro', 'journey-design-system', 'journey-visual-qa', 'journey-negatives', 'office-hours-brain-writeback', 'arm-benchmark-native-overbuild', 'arm-benchmark-crud-endpoint', 'arm-benchmark-bugfix-decoys', 'office-hours-section-loading', ]; const file = 'test/session-runner-stream-lifecycle.test.ts'; @@ -831,8 +821,7 @@ for (const file of ['test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures const selected = selectTests([file], E2E_TOUCHFILES); expect(selected.reason).toBe('diff'); expect(selected.selected.sort()).toEqual([ - 'auto-decide-preserved', 'autoplan-chain-pty', 'conductor-prose', - 'plan-ceo-finding-count', 'plan-ceo-mode-routing', 'plan-ceo-split-overflow', + 'auto-decide-preserved', 'autoplan-chain-pty', 'plan-ceo-finding-count', 'plan-ceo-mode-routing', 'plan-ceo-split-overflow', 'plan-design-finding-count', 'plan-design-review-plan-mode', 'plan-design-with-ui-scope', 'plan-devex-finding-count', 'plan-eng-finding-count', 'plan-eng-multi-finding-batching', 'plan-eng-review-plan-mode', @@ -844,7 +833,7 @@ for (const file of ['test/plan-count-cross-cwd-ancestry.test.ts', 'test/fixtures test('native clipped regressions retain the existing parser and owned-permission selection', () => { for (const [dependency, count, files] of [ - ['test/helpers/claude-pty-runner.ts', 22, [ + ['test/helpers/claude-pty-runner.ts', 20, [ 'test/plan-count-clipped-elision.test.ts', 'test/fixtures/eng-d1-clipped-elision-1579.json', 'test/fixtures/eng-d2-planning-prelude-4d.json', ]], ['test/helpers/plan-count-file-permission.ts', 10, [ diff --git a/test/plan-design-sdk-fixture.test.ts b/test/plan-design-sdk-fixture.test.ts index 1cc839213..b6aff6592 100644 --- a/test/plan-design-sdk-fixture.test.ts +++ b/test/plan-design-sdk-fixture.test.ts @@ -11,9 +11,9 @@ const ROOT = path.resolve(import.meta.dir, '..'); const source = fs.readFileSync(path.join(ROOT, 'test/skill-e2e-design.test.ts'), 'utf8'); const id = 'plan-design-review-plan-mode'; -// Same source-evaluation pattern as plan-tune-cathedral-fixture.test.ts: run -// the real suite registration and selected callback, without importing paid -// initialization. All filesystem mutations stay in this standalone fixture. +// Source-evaluation pattern: run the real suite registration and selected +// callback, without importing paid initialization. All filesystem mutations +// stay in this standalone fixture. async function exercise(mode: 'success' | 'max-turns' | 'first-timeout' | 'second-timeout' | 'saved-timeout' | 'empty-summary' | 'write-failure' | 'unchanged-seed' | 'no-additions' | 'short-plan' | 'api-error' | 'plan-read-failure' | 'attempt-deadline') { const scratch = fs.realpathSync(fs.mkdtempSync(path.join(os.tmpdir(), 'legacy-design-free-'))); const home = path.join(scratch, 'home'); fs.mkdirSync(home); diff --git a/test/plan-review-calibration.test.ts b/test/plan-review-calibration.test.ts index 64dee7b09..92bc0795f 100644 --- a/test/plan-review-calibration.test.ts +++ b/test/plan-review-calibration.test.ts @@ -26,12 +26,11 @@ test('semantic helper changes also select the separate DX analysis calibration', // Its direct behavioral consumers extend the unchanged helper-only set. if (file === 'test/plan-review-cases.test.ts') expected.push( 'plan-eng-review', 'plan-eng-review-artifact', 'plan-review-report', - 'plan-eng-review-plan-mode', 'plan-mode-no-op', 'conductor-prose', + 'plan-eng-review-plan-mode', 'plan-mode-no-op', 'carve-section-loading', 'autoplan-chain-pty', 'plan-eng-finding-floor', 'plan-eng-review-format-coverage', 'plan-eng-review-format-kind', 'plan-ceo-review-prosons-cadence', 'plan-review-prosons-format', - 'codex-offered-eng-review', 'codex-plan-eng-format-coverage', - 'codex-plan-eng-format-kind', 'plan-eng-coverage-audit', 'autoplan-dual-voice', + 'codex-offered-eng-review', 'plan-eng-coverage-audit', 'autoplan-dual-voice', ); expect(selectTests([file], E2E_TOUCHFILES, []).selected.sort()).toEqual( expected.sort()); diff --git a/test/plan-tune-cathedral-fixture.test.ts b/test/plan-tune-cathedral-fixture.test.ts deleted file mode 100644 index 0b59430b1..000000000 --- a/test/plan-tune-cathedral-fixture.test.ts +++ /dev/null @@ -1,129 +0,0 @@ -import { expect, test } from 'bun:test'; -import * as fs from 'node:fs'; -import * as path from 'node:path'; -import * as os from 'node:os'; -import { spawnSync } from 'node:child_process'; -import { E2E_TOUCHFILES } from './helpers/touchfiles'; - -const ROOT = path.resolve(import.meta.dir, '..'); -const source = fs.readFileSync(path.join(import.meta.dir, 'skill-e2e-plan-tune-cathedral.test.ts'), 'utf8'); -const names = ['plan-tune-hook-capture', 'plan-tune-enforcement', 'plan-tune-annotation', 'plan-tune-codex-import', 'plan-tune-dream-cycle']; - -async function exercise(selected = names, fault?: 'missing-log-lib' | 'missing-hook-lib' | 'first-hook' | 'setup', hostEnv: Record = {}) { - // Execute the actual selected callbacks with real local bins. Never import - // E2E initialization, call a provider, or pass ambient auth/host state. - const scratch = fs.mkdtempSync(path.join(os.tmpdir(), 'cathedral-contract-')); - const suites: any[] = [], finalizers: any[] = [], rows: any[] = [], attempts: any[] = [], dirs: string[] = [], removals: any[] = []; - const invoked: string[] = []; - let current: any, failedHook = false; - const owned = (p: string) => { if (!p.startsWith(scratch + path.sep)) throw new Error('Foreign fixture path'); }; - const env = { PATH: process.env.PATH, HOME: path.join(scratch, 'home'), GIT_CONFIG_NOSYSTEM: '1', GIT_CONFIG_GLOBAL: process.platform === 'win32' ? 'NUL' : '/dev/null', ...hostEnv }; - fs.mkdirSync(env.HOME); - const args: Record = { - ROOT, path, os: {tmpdir:()=>scratch}, process: {env}, expect, - beforeAll: (fn: any) => current.setups.push(fn), - afterAll: (fn: any) => current ? current.teardowns.push(fn) : finalizers.push(fn), - describeIfSelected: (_title: string, keys: string[], fn: any) => { - if (!keys.some(key => selected.includes(key))) return; - const suite = {setups:[],teardowns:[],tests:[]}; suites.push(suite); current=suite; fn(); current=undefined; - }, - testConcurrentIfSelected: (name: string, fn: any) => { if(selected.includes(name)) current.tests.push({name,fn}); }, - createEvalCollector: () => ({addTest:(row:any)=>rows.push(row)}), finalizeEvalCollector:()=>{}, - copyDirSync: (a: string, b: string) => { owned(b); fs.cpSync(a,b,{recursive:true}); }, - fs: { ...fs, - mkdtempSync(prefix: string) { owned(prefix); const dir=fs.mkdtempSync(prefix); dirs.push(dir); return dir; }, - copyFileSync(a: string,b: string) { - owned(b); - if(fault==='setup') throw new Error('Synthetic fixture copy failure'); - if((fault==='missing-log-lib' && a.endsWith('/lib/jsonl-store.ts')) || (fault==='missing-hook-lib' && a.endsWith('/lib/is-conductor.ts'))) return; - fs.copyFileSync(a,b); - }, - rmSync(p: string,opts: any) { - owned(p); const memory=path.join(p,'.gstack-state/free-text-memory.json'); - removals.push({path:p,nuggets:fs.existsSync(memory)?JSON.parse(fs.readFileSync(memory,'utf8')).nuggets.length:null}); - fs.rmSync(p,opts); - }, - }, - spawnSync: (bin: string,argv: string[],opts: any) => { - if(bin!=='git') { - owned(bin); - expect(['question-log-hook','question-preference-hook','gstack-codex-session-import','gstack-distill-apply']).toContain(path.basename(bin)); - owned(opts.env.GSTACK_STATE_ROOT); - } - if(opts.cwd) owned(opts.cwd); - expect(Number.isFinite(opts.timeout) && opts.timeout > 0).toBe(true); - invoked.push(path.basename(bin)); - if(fault==='first-hook' && path.basename(bin)==='question-preference-hook' && !failedHook) { - failedHook=true; return {status:1,stdout:'',stderr:'Synthetic transient hook failure'}; - } - return spawnSync(bin,argv,{...opts,timeout:opts.timeout,env:opts.env??env}); - }, - }; - let body=source; for(const m of source.matchAll(/^import[\s\S]*?;\n/gm)) body=body.replace(m[0],''); - try { - new Function(...Object.keys(args),new Bun.Transpiler({loader:'ts'}).transformSync(body))(...Object.values(args)); - for(const suite of suites) { - for(const setup of suite.setups) await setup(); - for(const callback of suite.tests) for(let attempt=1;attempt<=2;attempt++) { - let error: unknown; try {await callback.fn();} catch(e) {error=e;} - attempts.push({name:callback.name,attempt,error,remaining:dirs.filter(p=>fs.existsSync(p))}); - } - for(const done of suite.teardowns) await done(); - } - for(const done of finalizers) await done(); - return {rows,attempts,dirs,removals,invoked,allRemoved:dirs.every(p=>!fs.existsSync(p))}; - } finally {fs.rmSync(scratch,{recursive:true,force:true});} -} - -test('Cathedral callbacks run all five real local contracts with fresh attempts and truthful rows', async () => { - const x=await exercise(); - expect(x.attempts).toHaveLength(10); expect(x.rows).toHaveLength(10); expect(x.dirs).toHaveLength(10); - expect(new Set(x.dirs).size).toBe(10); expect(x.allRemoved).toBe(true); - for(const attempt of x.attempts) {expect(attempt.error).toBeUndefined();expect(attempt.remaining).toEqual([]);} - for(const row of x.rows) {expect(row.passed).toBe(true);expect(row.cost_usd).toBe(0);expect(row.output).toContain('no model invocation');} -}); - -test('Missing installed libraries still fail the actual hook and log contracts', async () => { - for(const [name,fault] of [['plan-tune-hook-capture','missing-log-lib'],['plan-tune-enforcement','missing-hook-lib']] as const) { - const x=await exercise([name],fault); - expect(x.rows).toHaveLength(2); expect(x.allRemoved).toBe(true); - for(const attempt of x.attempts) expect(attempt.error).toBeDefined(); - expect(x.rows.map(row=>row.passed)).toEqual([false,false]); - } -}); - -test('A failed dream hook retries with one fresh nugget and preserves the failed attempt', async () => { - const x=await exercise(['plan-tune-dream-cycle'],'first-hook'); - expect(x.attempts[0].error).toBeDefined(); expect(x.attempts[1].error).toBeUndefined(); - expect(x.rows.map(row=>row.passed)).toEqual([false,true]); - expect(x.removals.map(row=>row.nuggets)).toEqual([1,1]); - expect(new Set(x.dirs).size).toBe(2); expect(x.allRemoved).toBe(true); -}); - -test('A partial fixture setup failure is recorded once and cleans only its owned directory', async () => { - const x=await exercise(['plan-tune-hook-capture'],'setup'); - expect(x.rows.map(row=>row.passed)).toEqual([false,false]); expect(x.dirs).toHaveLength(2); expect(x.removals).toHaveLength(2); - expect(x.invoked.every(bin=>bin==='git')).toBe(true); expect(x.allRemoved).toBe(true); -}); - -test('Cathedral fixture controls and copied libraries select all five existing owners only', () => { - for(const file of ['test/plan-tune-cathedral-fixture.test.ts','lib/jsonl-store.ts','lib/is-conductor.ts']) { - const owners=Object.entries(E2E_TOUCHFILES).filter(([name,paths])=>names.includes(name)&&paths.includes(file)).map(([name])=>name); - expect(owners).toEqual(names); - } - expect(Object.entries(E2E_TOUCHFILES).filter(([,paths])=>paths.includes('test/plan-tune-cathedral-fixture.test.ts')).map(([name])=>name)).toEqual(names); -}); - - -test('Plain Claude cathedral contracts stay isolated from either inherited Conductor marker', async () => { - for (const hostEnv of [ - { CONDUCTOR_WORKSPACE_PATH: '/synthetic/conductor/workspace' }, - { CONDUCTOR_PORT: '55070', GSTACK_SESSION_KIND: 'spawned', OPENCLAW_SESSION: 'synthetic-session' }, - ]) { - const x = await exercise(['plan-tune-annotation', 'plan-tune-dream-cycle'], undefined, hostEnv); - expect(x.attempts).toHaveLength(4); - expect(x.rows.map(row => row.passed)).toEqual([true, true, true, true]); - for (const attempt of x.attempts) expect(attempt.error).toBeUndefined(); - expect(x.allRemoved).toBe(true); - } -}); diff --git a/test/skill-e2e-plan-tune-cathedral.test.ts b/test/plan-tune-cathedral.test.ts similarity index 90% rename from test/skill-e2e-plan-tune-cathedral.test.ts rename to test/plan-tune-cathedral.test.ts index f3f2e3a8d..4ac4fd8fb 100644 --- a/test/skill-e2e-plan-tune-cathedral.test.ts +++ b/test/plan-tune-cathedral.test.ts @@ -1,42 +1,22 @@ /** - * /plan-tune cathedral E2E (T16) — 5 scenarios, all gate tier per D12. + * /plan-tune cathedral contract tests (T16) — 5 scenarios, free suite. * * Each scenario verifies that the cathedral's substrate works end-to-end * through local hook and bin invocations. No model is called: these scenarios * exercise the installed-file contracts, using synthetic hook envelopes and * a synthetic Codex session. Unit tests cover the individual components. * - * Touchfile registration in test/helpers/touchfiles.ts: - * - plan-tune-hook-capture - * - plan-tune-enforcement - * - plan-tune-annotation - * - plan-tune-codex-import - * - plan-tune-dream-cycle - * * Each scenario uses GSTACK_STATE_ROOT to isolate from the user's real * ~/.gstack (per cathedral T1 + Codex D16 fix). Every attempt gets a fresh fixture. */ -import { afterAll, expect } from 'bun:test'; -import { - ROOT, - describeIfSelected, - testConcurrentIfSelected, - copyDirSync, - createEvalCollector, - finalizeEvalCollector, -} from './helpers/e2e-helpers'; +import { describe, expect, test } from 'bun:test'; +import { ROOT, copyDirSync } from './helpers/e2e-helpers'; import { spawnSync } from 'child_process'; import * as fs from 'fs'; import * as path from 'path'; import * as os from 'os'; -const collector = createEvalCollector('e2e-plan-tune-cathedral'); - -afterAll(() => { - finalizeEvalCollector(collector); -}); - /** Scaffold a fixture project with the bins + scripts the cathedral needs. */ function scaffoldFixture(workDir: string): { workDir: string; stateRoot: string; slug: string; env: NodeJS.ProcessEnv } { const stateRoot = path.join(workDir, '.gstack-state'); @@ -114,27 +94,18 @@ function cleanupFixture(workDir: string): void { } } -/** Own setup, assertions and cleanup inside each callback, including Bun retries. */ +/** Own setup, assertions and cleanup inside each callback. */ function testCathedral( name: string, prefix: string, run: (fixture: ReturnType) => Promise, ): void { - testConcurrentIfSelected(name, async () => { - const started = Date.now(); - let workDir: string | undefined; - let passed = false; + test(name, async () => { + const workDir = fs.mkdtempSync(path.join(os.tmpdir(), prefix)); try { - workDir = fs.mkdtempSync(path.join(os.tmpdir(), prefix)); await run(scaffoldFixture(workDir)); - passed = true; } finally { - if (workDir) cleanupFixture(workDir); - collector?.addTest({ - name, suite: 'plan-tune-cathedral', tier: 'e2e', passed, - duration_ms: Date.now() - started, cost_usd: 0, - output: 'Local hook/bin contract; no model invocation.', - }); + cleanupFixture(workDir); } }); } @@ -143,7 +114,7 @@ function testCathedral( // Scenario 1: Hook capture — PostToolUse hook writes to question-log.jsonl // --------------------------------------------------------------------------- -describeIfSelected('PlanTune cathedral E2E: hook capture', ['plan-tune-hook-capture'], () => { +describe('PlanTune cathedral E2E: hook capture', () => { testCathedral('plan-tune-hook-capture', 'cathedral-cap-', async (fixture) => { // Direct hook invocation simulates Claude Code's PostToolUse delivery. // E2E verifies the hook + bin chain works against real bins on disk @@ -190,7 +161,7 @@ describeIfSelected('PlanTune cathedral E2E: hook capture', ['plan-tune-hook-capt // Scenario 2: Enforcement — never-ask preference + marker + 2-way → deny // --------------------------------------------------------------------------- -describeIfSelected('PlanTune cathedral E2E: enforcement', ['plan-tune-enforcement'], () => { +describe('PlanTune cathedral E2E: enforcement', () => { testCathedral('plan-tune-enforcement', 'cathedral-enf-', async (fixture) => { fs.mkdirSync(path.join(fixture.stateRoot, 'projects', fixture.slug), { recursive: true }); fs.writeFileSync( @@ -253,7 +224,7 @@ describeIfSelected('PlanTune cathedral E2E: enforcement', ['plan-tune-enforcemen // Scenario 3: Annotation — declared profile injected via additionalContext // --------------------------------------------------------------------------- -describeIfSelected('PlanTune cathedral E2E: annotation', ['plan-tune-annotation'], () => { +describe('PlanTune cathedral E2E: annotation', () => { testCathedral('plan-tune-annotation', 'cathedral-ann-', async (fixture) => { // Strong declared profile that should annotate any signal_key=detail-preference question. fs.writeFileSync( @@ -318,7 +289,7 @@ describeIfSelected('PlanTune cathedral E2E: annotation', ['plan-tune-annotation' // Scenario 4: Codex import — JSONL session → import bin → log fills // --------------------------------------------------------------------------- -describeIfSelected('PlanTune cathedral E2E: codex import', ['plan-tune-codex-import'], () => { +describe('PlanTune cathedral E2E: codex import', () => { testCathedral('plan-tune-codex-import', 'cathedral-cdx-', async (fixture) => { const sessionFile = path.join(fixture.workDir, 'rollout-cathedral.jsonl'); const lines = [ @@ -374,7 +345,7 @@ describeIfSelected('PlanTune cathedral E2E: codex import', ['plan-tune-codex-imp // re-fire → memory injection // --------------------------------------------------------------------------- -describeIfSelected('PlanTune cathedral E2E: dream cycle', ['plan-tune-dream-cycle'], () => { +describe('PlanTune cathedral E2E: dream cycle', () => { testCathedral('plan-tune-dream-cycle', 'cathedral-dream-', async (fixture) => { // Seed proposals file directly (the SDK call is exercised by the unit // test; here we verify apply → re-fire round-trip on top of a known diff --git a/test/session-runner-groupkill.test.ts b/test/session-runner-groupkill.test.ts index 17a33d6c3..30741f5fd 100644 --- a/test/session-runner-groupkill.test.ts +++ b/test/session-runner-groupkill.test.ts @@ -87,14 +87,13 @@ describe('session-runner timeout kills the whole process group', () => { }, 60_000); }); -describe('all three provider runners carry the group-kill wiring', () => { - // Source pin, not behavior: codex/gemini need their real binaries for a +describe('both CLI provider runners carry the group-kill wiring', () => { + // Source pin, not behavior: codex needs its real binary for a // behavioral run, but the kill wiring is identical code — a runner that // drops `detached` or reverts to a bare kill() re-opens the orphan class. const runners = [ 'test/helpers/session-runner.ts', 'test/helpers/codex-session-runner.ts', - 'test/helpers/gemini-session-runner.ts', ]; for (const rel of runners) { test(`${path.basename(rel)}: detached spawn + killProcessGroup, no bare timeout kill`, () => { diff --git a/test/skill-e2e-brain-privacy-gate.test.ts b/test/skill-e2e-brain-privacy-gate.test.ts deleted file mode 100644 index 200a1c1f2..000000000 --- a/test/skill-e2e-brain-privacy-gate.test.ts +++ /dev/null @@ -1,233 +0,0 @@ -/** - * Privacy-gate E2E (periodic tier, paid). - * - * The gbrain-sync preamble block instructs the model to fire a one-time - * AskUserQuestion when: - * - `BRAIN_SYNC: off` in the preamble echo (sync mode not on) - * - config `artifacts_sync_mode_prompted` is "false" - * - gbrain is detected on the host (binary on PATH or `gbrain doctor` - * --fast --json succeeds) - * - * This test stages all three conditions (via env + a fake `gbrain` binary - * on PATH), runs a cheap gstack skill through the Agent SDK, intercepts - * every tool use via canUseTool, and asserts: one of the AskUserQuestions - * fired by the preamble is the privacy gate with its distinctive prose - * and three options (full / artifacts-only / decline). - * - * Cost: ~$0.30-$0.50 per run. Periodic tier (EVALS=1 EVALS_TIER=periodic). - * - * See scripts/resolvers/preamble/generate-brain-sync-block.ts for the - * prose contract this test locks in. - */ - -import { test, expect } from 'bun:test'; -import { CAPTURE_MS } from './helpers/eval-budgets'; -import { describeE2ETier } from './helpers/e2e-gate'; -import * as fs from 'fs'; -import * as os from 'os'; -import * as path from 'path'; -import { runAgentSdkTest, passThroughNonAskUserQuestion, resolveClaudeBinary } from './helpers/agent-sdk-runner'; - -const describeE2E = describeE2ETier('periodic'); - -describeE2E('gbrain-sync privacy gate fires once via preamble', () => { - test('gstack skill preamble fires the 3-option AskUserQuestion when gbrain is detected', async () => { - // Stage a fresh GSTACK_HOME with artifacts_sync_mode_prompted=false. - const gstackHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-gstack-')); - const fakeBinDir = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-bin-')); - // Fresh HOME with NO ~/.claude.json: on a machine where gbrain is - // registered type=http, the preamble's remote-mode detection reads the - // operator's ~/.claude.json and echoes "ARTIFACTS_SYNC: remote-mode" — - // and the local privacy gate legitimately never fires. An empty HOME - // makes the detection find nothing. - const tempHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-home-')); - - // Seed the config so the gate's condition passes. - fs.writeFileSync( - path.join(gstackHome, 'config.yaml'), - 'artifacts_sync_mode: off\nartifacts_sync_mode_prompted: false\n', - { mode: 0o600 } - ); - - // Fake `gbrain` binary that makes the host-detection probe succeed. - // The preamble checks `gbrain doctor --fast --json` OR `which gbrain`. - // Either branch counts as "gbrain detected." - fs.writeFileSync( - path.join(fakeBinDir, 'gbrain'), - '#!/bin/bash\n' + - 'case "$1" in\n' + - ' doctor) echo \'{"status":"ok","schema_version":2}\' ; exit 0 ;;\n' + - ' --version) echo "0.18.2" ; exit 0 ;;\n' + - ' *) exit 0 ;;\n' + - 'esac\n', - { mode: 0o755 } - ); - - const askUserQuestions: Array<{ input: Record }> = []; - const binary = resolveClaudeBinary(); - - // Per-test env, merged LAST by the hermetic env builder (safe post-v1.39: - // the runner always passes a COMPLETE hermetic env, so overrides can't - // break auth). Ambient process.env.GSTACK_HOME mutation does NOT work - // here — hermetic-env scrubs GSTACK_* and repoints GSTACK_HOME at its - // own singleton dir, so the staged config would never reach the child. - const childEnv = { - GSTACK_HOME: gstackHome, - HOME: tempHome, - PATH: `${fakeBinDir}:${process.env.PATH ?? '/usr/bin:/bin:/opt/homebrew/bin'}`, - }; - - try { - // Pick a small skill with the preamble and load it via Read to force - // the model to execute every preamble directive. A narrow "run /learn" - // prompt often gets reduced to a direct action, skipping the preamble - // gates. Mirror the plan-mode-no-op test pattern: ask the model to - // follow the skill's instructions in full. - const learnSkill = path.resolve( - import.meta.dir, - '..', - 'learn', - 'SKILL.md' - ); - await runAgentSdkTest({ - systemPrompt: { type: 'preset', preset: 'claude_code' }, - userPrompt: - `Read the skill file at ${learnSkill} and follow its instructions from the top, including every preamble directive. Execute every bash block. If any AskUserQuestion fires, present it.`, - workingDirectory: gstackHome, - maxTurns: 10, - allowedTools: ['Read', 'Grep', 'Glob', 'Bash'], - env: childEnv, - ...(binary ? { pathToClaudeCodeExecutable: binary } : {}), - canUseTool: async (toolName, input) => { - if (toolName === 'AskUserQuestion') { - askUserQuestions.push({ input }); - // Auto-answer "Decline — keep everything local" (option C) - // so the skill can continue without actually turning on sync. - const q = (input.questions as Array<{ - question: string; - options: Array<{ label: string }>; - }>)[0]; - const decline = - q.options.find((o) => /decline|keep everything local|no thanks/i.test(o.label)) ?? - q.options[q.options.length - 1]!; - return { - behavior: 'allow', - updatedInput: { - questions: input.questions, - answers: { [q.question]: decline.label }, - }, - }; - } - return passThroughNonAskUserQuestion(toolName, input); - }, - }); - - // Assertion 1: the privacy gate fired. - const privacyQuestions = askUserQuestions.filter((aq) => { - const qs = aq.input.questions as Array<{ question: string }>; - return qs.some( - (q) => - /publish.*session memory|private github repo|gbrain indexes/i.test(q.question) - ); - }); - expect(privacyQuestions.length).toBeGreaterThanOrEqual(1); - - // Assertion 2: the question has the three expected options. - const gate = privacyQuestions[0]!.input.questions as Array<{ - question: string; - options: Array<{ label: string }>; - }>; - const labels = gate[0]!.options.map((o) => o.label.toLowerCase()).join(' | '); - // Full / artifacts-only / decline are the three canonical options. - expect(labels).toMatch(/everything|allowlisted|full/); - expect(labels).toMatch(/artifact/); - expect(labels).toMatch(/decline|local|no thanks/); - - // Assertion 3: the gate should NOT fire twice in one run. - // (The preamble is supposed to be idempotent within a session.) - expect(privacyQuestions.length).toBe(1); - } finally { - fs.rmSync(gstackHome, { recursive: true, force: true }); - fs.rmSync(fakeBinDir, { recursive: true, force: true }); - fs.rmSync(tempHome, { recursive: true, force: true }); - } - }, CAPTURE_MS); - - test('privacy gate does NOT fire when artifacts_sync_mode_prompted is already true', async () => { - // Same staging, but prompted=true this time. Gate should be silent. - const gstackHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-off-')); - const fakeBinDir = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-off-bin-')); - // Fresh HOME without a .claude.json — same rationale as the first test: - // without it the operator's ~/.claude.json flips the preamble into - // remote-mode and this negative test passes vacuously. - const tempHome = fs.mkdtempSync(path.join(os.tmpdir(), 'privacy-gate-off-home-')); - - fs.writeFileSync( - path.join(gstackHome, 'config.yaml'), - 'artifacts_sync_mode: off\nartifacts_sync_mode_prompted: true\n', - { mode: 0o600 } - ); - - fs.writeFileSync( - path.join(fakeBinDir, 'gbrain'), - '#!/bin/bash\necho \'{"status":"ok"}\'\nexit 0\n', - { mode: 0o755 } - ); - - const askUserQuestions: Array<{ input: Record }> = []; - const binary = resolveClaudeBinary(); - - // Per-test env, merged LAST by the hermetic env builder (see note on the - // first test — ambient GSTACK_HOME mutation is scrubbed by hermetic-env). - const childEnv = { - GSTACK_HOME: gstackHome, - HOME: tempHome, - PATH: `${fakeBinDir}:${process.env.PATH ?? '/usr/bin:/bin:/opt/homebrew/bin'}`, - }; - - try { - await runAgentSdkTest({ - systemPrompt: { type: 'preset', preset: 'claude_code' }, - userPrompt: - 'Run /learn with no arguments. Just report the learnings count.', - workingDirectory: gstackHome, - maxTurns: 4, - allowedTools: ['Read', 'Grep', 'Glob', 'Bash'], - env: childEnv, - ...(binary ? { pathToClaudeCodeExecutable: binary } : {}), - canUseTool: async (toolName, input) => { - if (toolName === 'AskUserQuestion') { - askUserQuestions.push({ input }); - // Pass through whatever the model asks; don't prefer anything. - const q = (input.questions as Array<{ - question: string; - options: Array<{ label: string }>; - }>)[0]; - return { - behavior: 'allow', - updatedInput: { - questions: input.questions, - answers: { [q.question]: q.options[0]!.label }, - }, - }; - } - return passThroughNonAskUserQuestion(toolName, input); - }, - }); - - // No AskUserQuestion should have matched the privacy gate's prose. - const privacyQuestions = askUserQuestions.filter((aq) => { - const qs = aq.input.questions as Array<{ question: string }>; - return qs.some( - (q) => - /publish.*session memory|private github repo|gbrain indexes/i.test(q.question) - ); - }); - expect(privacyQuestions.length).toBe(0); - } finally { - fs.rmSync(gstackHome, { recursive: true, force: true }); - fs.rmSync(fakeBinDir, { recursive: true, force: true }); - fs.rmSync(tempHome, { recursive: true, force: true }); - } - }, CAPTURE_MS); -}); diff --git a/test/skill-e2e-conductor-prose.test.ts b/test/skill-e2e-conductor-prose.test.ts deleted file mode 100644 index 33337bd1b..000000000 --- a/test/skill-e2e-conductor-prose.test.ts +++ /dev/null @@ -1,76 +0,0 @@ -/** - * Conductor → prose decision brief (periodic-tier, paid, real-PTY). - * - * Proves the end-to-end behavior: when CONDUCTOR_SESSION is signalled, a skill - * that hits a decision renders a PROSE decision brief and waits, instead of - * silently skipping the user. - * - * SCOPE — read before trusting this as the Conductor guard. This is END-TO-END - * BEHAVIOR coverage, NOT the discriminating Conductor guarantee: - * - The deterministic guard is test/question-preference-hook.test.ts - * ("Conductor prose redirect") — it sets process.env.CONDUCTOR_* and asserts - * the PreToolUse hook denies + redirects. That test CAN fail on unfixed code. - * - The PTY harness here cannot register `mcp__conductor__AskUserQuestion`, so - * it tests "native AUQ unavailable + Conductor signal → prose," NOT "the MCP - * variant exists and must not be called" (Codex #10). Under --disallowedTools - * a present-human interactive session already prose-falls-back, so this test - * is a smoke check that the Conductor path still produces a prose brief, not - * a proof that the Conductor signal (vs the generic fallback) drove it. - * - * Periodic tier: model-behavior, non-deterministic. - */ - -import { test, expect } from 'bun:test'; -import { CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets'; -import { describeE2ETier } from './helpers/e2e-gate'; -import { runPlanSkillObservation } from './helpers/claude-pty-runner'; - -const describeE2E = describeE2ETier('periodic'); - -const FLAWED_PLAN = `# Plan: add a "developer-friendly" pricing tier - -## Goal -Increase developer adoption. - -## Premise -No tests mentioned, no rollout plan, no auth check on the upgrade endpoint. -Adds a Stripe tier, a React pricing page, a Postgres entitlements table, and a -Redis cache. The team "feels like" it should be cheaper; no developer was asked. -`; - -describeE2E('Conductor renders decisions as prose (periodic)', () => { - test('plan-eng-review in a Conductor session surfaces a PROSE decision brief, not a silent skip', async () => { - const obs = await runPlanSkillObservation({ - skillName: 'plan-eng-review', - inPlanMode: true, - // Mimic Conductor: native AUQ disabled + the Conductor env signal present. - extraArgs: ['--disallowedTools', 'AskUserQuestion'], - env: { CONDUCTOR_WORKSPACE_PATH: '/tmp/conductor-prose-e2e' }, - initialPlanContent: FLAWED_PLAN, - // A judge can mistake a streaming partial option for a waiting brief. - // Keep observing until the independent prose evidence is available. - requireProseEvidence: true, - timeoutMs: CAPTURE_MS, - }); - - // The decision must reach the human as prose. 'silent_write' (wrote findings - // to the plan without asking) is the precise failure we guard against. - if (obs.outcome === 'silent_write') { - throw new Error( - `Conductor prose regression: skill wrote findings without surfacing a decision.\n` + - `summary: ${obs.summary}\n--- evidence ---\n${obs.evidence}`, - ); - } - if (obs.outcome === 'exited' || obs.outcome === 'timeout') { - throw new Error( - `Conductor prose test inconclusive: outcome=${obs.outcome}\n` + - `summary: ${obs.summary}\n--- evidence ---\n${obs.evidence}`, - ); - } - // A prose-rendered decision brief was observed at some point in the run. - expect(obs.proseAUQEverObserved, - `Conductor prose decision not observed: outcome=${obs.outcome}\n` + - `summary: ${obs.summary}\n--- evidence ---\n${obs.evidence}`, - ).toBe(true); - }, CAPTURE_LONG_MS); -}); diff --git a/test/skill-e2e-opus-47.test.ts b/test/skill-e2e-opus-47.test.ts deleted file mode 100644 index 70c669c5d..000000000 --- a/test/skill-e2e-opus-47.test.ts +++ /dev/null @@ -1,268 +0,0 @@ -/** - * Opus 4.7 behavior evals. - * - * One case, pinned to claude-opus-4-7: - * - * Routing precision — the "when in doubt, invoke the skill" policy should - * route ambiguous dev prompts to the right skill WITHOUT routing - * casual/non-dev prompts. A handful of positive and negative controls. - * (The fanout A/B retired 2026-08 — single-run parallel-call comparison was - * a coin flip; the SDK overlay-harness is the maintained instrument.) - * - * Both cases require a running Anthropic API key. Gated behind EVALS=1. - * Classify as `periodic` in touchfiles — behavior measurement, not gate. - */ - -import { describe, test, expect, afterAll } from 'bun:test'; -import { JUDGE_MS, CAPTURE_MS, CAPTURE_LONG_MS } from './helpers/eval-budgets'; -import { runSkillTest } from './helpers/session-runner'; -import { EvalCollector } from './helpers/eval-store'; -import { extractSkillHead } from './helpers/skill-fixture'; -import { spawnSync } from 'child_process'; -import * as fs from 'fs'; -import * as path from 'path'; -import * as os from 'os'; - -const ROOT = path.resolve(import.meta.dir, '..'); -const OPUS_47 = 'claude-opus-4-7'; - -const evalsEnabled = !!process.env.EVALS; -const describeE2E = evalsEnabled ? describe : describe.skip; -const evalCollector = evalsEnabled ? new EvalCollector('e2e-opus-47') : null; -const runId = new Date().toISOString().replace(/[:.]/g, '').replace('T', '-').slice(0, 15); - -// --- Helpers --- - -/** Skills that must exist as individual .claude/skills/{name}/SKILL.md files - * for Claude Code's auto-discovery to treat them as invokable via Skill tool. - * Matches the pattern in skill-routing-e2e.test.ts. */ -const INSTALLED_SKILLS = [ - 'qa', 'qa-only', 'ship', 'review', 'plan-ceo-review', 'plan-eng-review', - 'plan-design-review', 'design-review', 'design-consultation', 'retro', - 'document-release', 'investigate', 'office-hours', 'browse', -]; - -/** Write a scratch root with: - * - Per-skill SKILL.md files under .claude/skills/ (so Skill tool sees them) - * - Project CLAUDE.md with explicit routing rules AND (optionally) the - * 4.7 overlay content directly inlined so `claude -p` sees it - * - git init - * - * `includeOverlay` controls whether the opus-4-7 nudges (Fan out, Literal, - * etc.) get inlined into CLAUDE.md — this is the A/B axis for the fanout - * test. `claude -p` doesn't auto-load SKILL.md content, so CLAUDE.md is - * the only way to make the overlay visible to the model in this test - * harness. - */ -function mkEvalRoot(suffix: string, includeOverlay: boolean): string { - const tmp = fs.mkdtempSync(path.join(os.tmpdir(), `opus47-${suffix}-`)); - - // Render at opus-4-7 INTO A MKDTEMP via --out-dir — never the live tree. - // The previous cwd=ROOT regeneration rewrote every in-repo SKILL.md - // mid-run while concurrent paid shards copyFileSync those same files in - // their beforeAll (EVALS_JOBS>=4 locally, 2 per CI slice): a sibling could - // capture a half-regenerated or opus-rendered SKILL.md, and a timeout - // before afterAll left the whole tree rendered at the wrong model for - // every later shard. --out-dir mirrors the repo layout (// - // SKILL.md), which is all this fixture reads. - const renderDir = fs.mkdtempSync(path.join(os.tmpdir(), `opus47-render-${suffix}-`)); - const result = spawnSync( - 'bun', - ['run', 'scripts/gen-skill-docs.ts', '--model', includeOverlay ? 'opus-4-7' : 'claude', '--out-dir', renderDir], - { cwd: ROOT, stdio: 'pipe', encoding: 'utf-8', timeout: 60_000 }, - ); - if (result.status !== 0) { - throw new Error(`gen-skill-docs failed: ${result.stderr}`); - } - - // Install per-skill SKILL.md files for Skill tool discovery. Routing only - // reads the frontmatter (name + description), so install frontmatter + the - // first ~30 body lines instead of the full 1000-1900-line files - // (CLAUDE.md: "E2E test fixtures: extract, don't copy"). - const skillsDir = path.join(tmp, '.claude', 'skills'); - for (const skill of INSTALLED_SKILLS) { - const src = path.join(renderDir, skill, 'SKILL.md'); - if (!fs.existsSync(src)) continue; - const destDir = path.join(skillsDir, skill); - fs.mkdirSync(destDir, { recursive: true }); - fs.writeFileSync(path.join(destDir, 'SKILL.md'), extractSkillHead(src)); - } - fs.rmSync(renderDir, { recursive: true, force: true }); - - // Extract the opus-4-7 model-overlay content from the checked-in file - // so we can inline it into CLAUDE.md when includeOverlay is true. - const overlayText = includeOverlay - ? fs.readFileSync(path.join(ROOT, 'model-overlays', 'opus-4-7.md'), 'utf-8') - .replace(/\{\{INHERIT:claude\}\}\s*/, '') - .trim() - : ''; - - // Project CLAUDE.md. Explicit routing rules so the agent reaches for - // Skill tool on matching prompts, plus the optional overlay. - const routingBlock = `## Skill routing - -When the user's request matches an available skill, invoke it via the Skill tool -as your FIRST action. The skill has multi-step workflows, checklists, and quality -gates that produce better results than an ad-hoc answer. When in doubt, invoke. - -- Bugs, errors, "why is this broken", "wtf" → invoke investigate -- Ship, deploy, "send it", create a PR → invoke ship -- QA, test the site, "does this work" → invoke qa -- Code review, check my diff → invoke review -- Product ideas, brainstorming, "is this worth building" → invoke office-hours -- Architecture, "does this design make sense" → invoke plan-eng-review -- Design system, visual polish → invoke design-review -- Weekly retro, what did we ship → invoke retro`; - - const claudeMd = includeOverlay - ? `# Project\n\n${overlayText}\n\n${routingBlock}\n` - : `# Project\n\n${routingBlock}\n`; - - fs.writeFileSync(path.join(tmp, 'CLAUDE.md'), claudeMd); - fs.writeFileSync(path.join(tmp, 'package.json'), '{"name":"opus47-eval"}'); - - const git = (args: string[]) => - spawnSync('git', args, { cwd: tmp, stdio: 'pipe', timeout: 5_000 }); - git(['init']); - git(['config', 'user.email', 't@t.com']); - git(['config', 'user.name', 'T']); - git(['add', '.']); - git(['commit', '-m', 'init']); - - return tmp; -} - -/** Count parallel tool calls in the first assistant turn. */ -function firstTurnParallelism(transcript: any[]): number { - const firstAssistant = transcript.find((e) => e.type === 'assistant'); - if (!firstAssistant) return 0; - const content = firstAssistant.message?.content ?? []; - return content.filter((c: any) => c.type === 'tool_use').length; -} - -interface RoutingCase { - name: string; - prompt: string; - shouldRoute: boolean; - expectedSkill?: string; -} - -/** Small, intentionally chosen routing cases. Positive cases are ambiguous - * phrasings the user actually says, not template text. Negative cases are - * casual or off-topic prompts that match routing keywords but shouldn't - * trigger a skill. */ -const ROUTING_CASES: RoutingCase[] = [ - // Positive — should route - { name: 'pos-wtf-bug', prompt: "wtf is this error coming from auth.ts:47 when the cookie expires?", shouldRoute: true, expectedSkill: 'investigate' }, - { name: 'pos-send-it', prompt: "ok this is good enough, let's send it.", shouldRoute: true, expectedSkill: 'ship' }, - { name: 'pos-does-it-work', prompt: "I just pushed the login flow changes. Test the deployed site and find any bugs.", shouldRoute: true, expectedSkill: 'qa' }, - // Negative — should NOT route - { name: 'neg-syntax-q', prompt: "wtf does this Python list comprehension syntax even mean, [x for x in y if z]?", shouldRoute: false }, - { name: 'neg-algo-q', prompt: "does this bubble sort algorithm actually work in O(n log n)?", shouldRoute: false }, - { name: 'neg-slack-send', prompt: "can you help me write the slack message? I want to send it to the team.", shouldRoute: false }, -]; - -// --- Tests --- - -describeE2E('Opus 4.7 overlay behavior evals', () => { - afterAll(() => { - evalCollector?.finalize(); - // No tree restore needed: mkEvalRoot renders into a mkdtemp via - // --out-dir, so the live repo's SKILL.md files are never touched — a - // timeout mid-run can no longer strand the tree at the wrong model for - // concurrent shards. - }); - - // (fanout A/B retired, 2026-08 audit: it compared parallel-call counts of - // two SINGLE stochastic runs — parA >= parB is a coin flip with near-zero - // remaining information; the overlay-fanout question is answered and the - // SDK overlay-harness (test/skill-e2e-overlay-harness.test.ts) is the - // maintained instrument for the next experiment.) - - - test( - 'routing precision: positives route, negatives do not', - async () => { - // Single SKILL.md tree shared by all cases. We run claude-opus-4-7 with - // tool access to Skill; measure whether the first tool call is Skill(..) - // and if so, which skill. - const root = mkEvalRoot('routing', true); - - try { - const results = await Promise.all( - ROUTING_CASES.map((c) => - runSkillTest({ - prompt: c.prompt, - workingDirectory: root, - maxTurns: 3, - allowedTools: ['Skill', 'Read', 'Bash', 'Glob', 'Grep'], - timeout: JUDGE_MS, - testName: `routing-${c.name}`, - runId, - model: OPUS_47, - }).then((r) => ({ c, r })), - ), - ); - - let tp = 0, fn = 0, fp = 0, tn = 0; - const rows: string[] = []; - let totalCost = 0; - - for (const { c, r } of results) { - const skillCalls = r.toolCalls.filter((tc) => tc.tool === 'Skill'); - const routed = skillCalls.length > 0; - const actualSkill = routed ? skillCalls[0]?.input?.skill : undefined; - - const correct = c.shouldRoute - ? routed && (!c.expectedSkill || actualSkill === c.expectedSkill) - : !routed; - - if (c.shouldRoute && routed) tp++; - else if (c.shouldRoute && !routed) fn++; - else if (!c.shouldRoute && routed) fp++; - else tn++; - - totalCost += r.costEstimate.estimatedCost; - rows.push( - ` ${c.name.padEnd(18)} routed=${String(routed).padEnd(5)} skill=${String(actualSkill).padEnd(16)} ` + - `expected=${c.shouldRoute ? (c.expectedSkill ?? 'any') : '(none)'} ${correct ? 'OK' : 'MISS'}`, - ); - - evalCollector?.addTest({ - name: `routing-${c.name}`, - suite: 'Opus 4.7 routing', - tier: 'e2e', - passed: correct, - duration_ms: r.duration, - cost_usd: r.costEstimate.estimatedCost, - transcript: r.transcript, - output: `routed=${routed} actual=${actualSkill ?? '(none)'} expected=${c.shouldRoute ? c.expectedSkill ?? 'any' : '(none)'}`, - turns_used: r.costEstimate.turnsUsed, - exit_reason: r.exitReason, - }); - } - - const posCount = ROUTING_CASES.filter((c) => c.shouldRoute).length; - const negCount = ROUTING_CASES.length - posCount; - const tpRate = posCount > 0 ? tp / posCount : 0; - const fpRate = negCount > 0 ? fp / negCount : 0; - - console.log(`[opus-4-7 routing] total cost $${totalCost.toFixed(2)}`); - console.log(rows.join('\n')); - console.log( - ` TP=${tp}/${posCount} (${(tpRate * 100).toFixed(0)}%) FN=${fn} ` + - `FP=${fp}/${negCount} (${(fpRate * 100).toFixed(0)}%) TN=${tn}`, - ); - - // Thresholds from the test plan artifact: TP >= 80%, FP <= 30%. - // With a small N we loosen slightly: TP >= 66% (2 of 3 positive), - // FP <= 33% (no more than 1 of 3 negatives). - expect(tpRate, `true-positive rate ${(tpRate * 100).toFixed(0)}% (need >= 66%)`).toBeGreaterThanOrEqual(2 / 3); - expect(fpRate, `false-positive rate ${(fpRate * 100).toFixed(0)}% (need <= 33%)`).toBeLessThanOrEqual(1 / 3); - } finally { - fs.rmSync(root, { recursive: true, force: true }); - } - }, - CAPTURE_LONG_MS, - ); -}); diff --git a/test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial-sonnet.test.ts b/test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial-sonnet.test.ts deleted file mode 100644 index ef95b8c1d..000000000 --- a/test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial-sonnet.test.ts +++ /dev/null @@ -1,6 +0,0 @@ -import { describeE2ETier } from './helpers/e2e-gate'; -import { registerOverlayCase } from './helpers/overlay-case'; - -describeE2ETier('periodic')('overlay behavior contract v2 (SDK)', () => { - registerOverlayCase('opus-4-7-effort-match-trivial-sonnet'); -}); diff --git a/test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation-sonnet.test.ts b/test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation-sonnet.test.ts deleted file mode 100644 index d7806cd0f..000000000 --- a/test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation-sonnet.test.ts +++ /dev/null @@ -1,6 +0,0 @@ -import { describeE2ETier } from './helpers/e2e-gate'; -import { registerOverlayCase } from './helpers/overlay-case'; - -describeE2ETier('periodic')('overlay behavior contract v2 (SDK)', () => { - registerOverlayCase('opus-4-7-literal-interpretation-sonnet'); -}); diff --git a/test/skill-e2e-ship-idempotency.test.ts b/test/skill-e2e-ship-idempotency.test.ts deleted file mode 100644 index 29ef83827..000000000 --- a/test/skill-e2e-ship-idempotency.test.ts +++ /dev/null @@ -1,285 +0,0 @@ -/** - * /ship idempotency E2E (periodic, paid, real-PTY). - * - * Asserts: when /ship runs against a branch that has ALREADY been bumped - * (VERSION ahead of base AND package.json synced AND a CHANGELOG entry - * exists for the bumped version), the workflow: - * - * 1. Detects ALREADY_BUMPED state via the Step 12 idempotency check - * 2. Does NOT echo STATE: FRESH (which would trigger a second bump) - * 3. Does NOT mutate the fixture's VERSION file - * 4. Does NOT append a duplicate CHANGELOG [0.0.2] entry - * 5. Does NOT create a new "chore: bump version" commit - * - * Why real-PTY: the old SDK-harness ship-idempotency variant (removed in - * v1.64.1.0 as redundant with this test) used a synthetic prompt asking - * the agent to "run ONLY the idempotency checks." This test exercises the - * actual /ship skill end-to-end against a real git fixture so a regression - * that silently re-bumps despite the check passing would be caught. - * - * Plan-mode framing: we run /ship in plan mode so the agent cannot push, - * commit, or open PRs. The Step 12 idempotency check is read-only - * (reads VERSION + package.json + git rev-parse) and runs fine in plan - * mode. The plan-ready output serves as the terminal signal — the agent - * has done its analysis and produced a plan describing what it would do. - * - * If the agent decides to bump or push despite the fixture's - * ALREADY_BUMPED state, that intent surfaces in the plan or in - * tool-call attempts, which we detect. - * - * Cost: ~$2-4/run. Periodic tier — long, runs weekly. - */ - -import { test, expect } from 'bun:test'; -import { PTY_LONG_MS } from './helpers/eval-budgets'; -import { describeE2ETier } from './helpers/e2e-gate'; -import { spawnSync } from 'child_process'; -import * as fs from 'fs'; -import * as path from 'path'; -import * as os from 'os'; -import { - launchClaudePty, - isPermissionDialogVisible, - isNumberedOptionListVisible, -} from './helpers/claude-pty-runner'; - -const describeE2E = describeE2ETier('periodic'); - -interface ShipFixture { - workTree: string; - bareRemote: string; - /** Full bash log of `git` and helper commands run during setup. */ - setupLog: string[]; -} - -/** - * Build a self-contained git fixture representing an already-shipped state: - * - main branch at VERSION 0.0.1, with one CHANGELOG entry [0.0.1] - * - feat/already-shipped branch at VERSION 0.0.2 (bumped + synced), - * CHANGELOG has [0.0.2] entry on top of [0.0.1], one feature commit - * - bareRemote is the origin; both branches are pushed - * - * Returns the work-tree dir for /ship to operate on. - */ -function buildShippedFixture(): ShipFixture { - const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-ship-fixture-')); - const workTree = path.join(root, 'workspace'); - const bareRemote = path.join(root, 'origin.git'); - fs.mkdirSync(workTree, { recursive: true }); - - const setupLog: string[] = []; - const sh = (cmd: string, cwd: string): void => { - setupLog.push(`[${cwd}] ${cmd}`); - const result = spawnSync('bash', ['-c', cmd], { cwd, stdio: 'pipe', timeout: 15_000 }); - if (result.status !== 0) { - const stderr = result.stderr?.toString() ?? ''; - throw new Error(`fixture setup failed at "${cmd}":\n${stderr}\n--- log ---\n${setupLog.join('\n')}`); - } - }; - - // Bare remote. - sh(`git init --bare "${bareRemote}"`, root); - - // Initial commit on main. - sh('git init -b main', workTree); - sh('git config user.email "test@test.com"', workTree); - sh('git config user.name "Test"', workTree); - sh('git config commit.gpgsign false', workTree); - - fs.writeFileSync(path.join(workTree, 'VERSION'), '0.0.1\n'); - fs.writeFileSync( - path.join(workTree, 'package.json'), - JSON.stringify({ name: 'fixture', version: '0.0.1', private: true }, null, 2) + '\n', - ); - fs.writeFileSync( - path.join(workTree, 'CHANGELOG.md'), - `# Changelog\n\n## [0.0.1] - 2026-01-01\n\n- Initial release\n`, - ); - fs.writeFileSync(path.join(workTree, 'README.md'), '# Fixture\n'); - - sh('git add VERSION package.json CHANGELOG.md README.md', workTree); - sh('git commit -m "chore: initial release v0.0.1"', workTree); - sh(`git remote add origin "${bareRemote}"`, workTree); - sh('git push -u origin main', workTree); - - // Feature branch with ALREADY_BUMPED state. - sh('git checkout -b feat/already-shipped', workTree); - fs.writeFileSync(path.join(workTree, 'VERSION'), '0.0.2\n'); - fs.writeFileSync( - path.join(workTree, 'package.json'), - JSON.stringify({ name: 'fixture', version: '0.0.2', private: true }, null, 2) + '\n', - ); - fs.writeFileSync( - path.join(workTree, 'CHANGELOG.md'), - `# Changelog\n\n## [0.0.2] - 2026-04-25\n\n**Feature shipped.**\n\nAdded the new feature.\n\n## [0.0.1] - 2026-01-01\n\n- Initial release\n`, - ); - fs.writeFileSync(path.join(workTree, 'feature.md'), '# Feature\n\nAlready shipped.\n'); - - sh('git add VERSION package.json CHANGELOG.md feature.md', workTree); - sh('git commit -m "feat: add new feature\n\nbumps VERSION to 0.0.2"', workTree); - sh('git push -u origin feat/already-shipped', workTree); - - return { workTree, bareRemote, setupLog }; -} - -/** Snapshot the load-bearing fixture state so we can compare post-run. */ -interface FixtureSnapshot { - versionFile: string; - packageVersion: string; - changelogEntryCount: number; - bumpCommitCount: number; - branchHead: string; -} - -function snapshotFixture(workTree: string): FixtureSnapshot { - const versionFile = fs.readFileSync(path.join(workTree, 'VERSION'), 'utf-8').trim(); - const pkg = JSON.parse(fs.readFileSync(path.join(workTree, 'package.json'), 'utf-8')); - const changelog = fs.readFileSync(path.join(workTree, 'CHANGELOG.md'), 'utf-8'); - // Count `## [0.0.2]` headings — should stay at 1 across re-runs. - const changelogEntryCount = (changelog.match(/^##\s*\[0\.0\.2\]/gm) ?? []).length; - const head = spawnSync('git', ['rev-parse', 'HEAD'], { cwd: workTree, stdio: 'pipe', timeout: 30_000 }); - const branchHead = head.stdout?.toString().trim() ?? ''; - // Count "chore: bump version" commits on this branch since main. - const log = spawnSync( - 'git', ['log', '--format=%s', 'main..HEAD'], - { cwd: workTree, stdio: 'pipe', timeout: 30_000 }, - ); - const subjects = log.stdout?.toString() ?? ''; - const bumpCommitCount = subjects.split('\n').filter(s => /chore:\s*bump\s+version/i.test(s)).length; - return { versionFile, packageVersion: pkg.version, changelogEntryCount, bumpCommitCount, branchHead }; -} - -describeE2E('/ship idempotency E2E (periodic, real-PTY)', () => { - test( - 'rerunning /ship on an already-shipped branch detects ALREADY_BUMPED and does not mutate fixture', - async () => { - const fixture = buildShippedFixture(); - const before = snapshotFixture(fixture.workTree); - - const session = await launchClaudePty({ - permissionMode: 'plan', - cwd: fixture.workTree, - timeoutMs: PTY_LONG_MS, - // Disable network-y pieces so the agent can't reach actual github. - env: { GH_TOKEN: 'mock-not-real', NO_COLOR: '1' }, - seedSkills: true, - }); - - let outcome: 'detected' | 'plan_ready' | 'attempted_mutation' | 'timeout' | 'exited' = 'timeout'; - let evidence = ''; - - try { - await Bun.sleep(8000); - const since = session.mark(); - session.send('/ship\r'); - - const budgetMs = 900_000; - const start = Date.now(); - let lastPermSig = ''; - while (Date.now() - start < budgetMs) { - await Bun.sleep(3000); - if (session.exited()) { - outcome = 'exited'; - evidence = session.visibleSince(since).slice(-3000); - break; - } - const visible = session.visibleSince(since); - - // Auto-grant any permission dialogs the preamble triggers - // (e.g. touch on a marker file claude considers sensitive). - // Classify on the recent tail; don't double-press the same render. - const tail = visible.slice(-1500); - if (isNumberedOptionListVisible(tail) && isPermissionDialogVisible(tail)) { - const sig = visible.slice(-500); - if (sig !== lastPermSig) { - lastPermSig = sig; - session.send('1\r'); - await Bun.sleep(1500); - continue; - } - } - - // Positive: idempotency classify reported ALREADY_BUMPED. Post-carve - // (T9), Step 12 runs `gstack-version-bump classify` which emits JSON - // (`"state":"ALREADY_BUMPED"`); the legacy inline bash echoed - // `STATE: ALREADY_BUMPED`. Accept either so the test survives the carve. - if (/STATE:\s*ALREADY_BUMPED|"state":\s*"ALREADY_BUMPED"/.test(visible)) { - outcome = 'detected'; - evidence = visible.slice(-3000); - break; - } - - // Negative regressions: - // - classify reported FRESH (CLI JSON or legacy echo) → would re-bump - // - agent attempted git commit -m "chore: bump version" - // - agent attempted git push - // - agent ran the CLI write path (gstack-version-bump write) — a - // re-bump on an already-shipped branch - if ( - /"state":\s*"FRESH"/.test(visible) || - /STATE:\s*FRESH(?![\w-])/i.test(visible) || - /gstack-version-bump\s+write/i.test(visible) || - /git\s+commit\s+.*chore:\s*bump\s+version/i.test(visible) || - /git\s+push.*origin/i.test(visible) - ) { - outcome = 'attempted_mutation'; - evidence = visible.slice(-3000); - break; - } - - // Plan-ready outcome (acceptable terminal): the agent finished - // analysis. We'll accept this if no mutation signals showed up. - if (/ready to execute|Would you like to proceed/i.test(visible)) { - outcome = 'plan_ready'; - evidence = visible.slice(-3000); - break; - } - } - // Budget exhausted without a terminal signal: capture the tail NOW, - // while the session is still alive. Only the break paths above set - // evidence — without this, the timeout throw ships evidence: "". - if (outcome === 'timeout') { - evidence = session.visibleSince(since).slice(-3000); - } - } finally { - await session.close(); - } - - // Verify fixture was not mutated regardless of outcome. - const after = snapshotFixture(fixture.workTree); - const fixtureStable = - after.versionFile === before.versionFile && - after.packageVersion === before.packageVersion && - after.changelogEntryCount === before.changelogEntryCount && - after.bumpCommitCount === before.bumpCommitCount && - after.branchHead === before.branchHead; - - try { - if (outcome === 'attempted_mutation') { - throw new Error( - `/ship attempted to mutate already-shipped state.\n` + - `--- evidence (last 3KB) ---\n${evidence}\n` + - `--- before ---\n${JSON.stringify(before, null, 2)}\n` + - `--- after ---\n${JSON.stringify(after, null, 2)}`, - ); - } - if (outcome === 'exited') { - throw new Error(`claude exited unexpectedly.\n--- evidence ---\n${evidence}`); - } - if (outcome === 'timeout') { - throw new Error( - `Timed out before any terminal outcome.\n--- evidence (last 3KB) ---\n${evidence}`, - ); - } - // Detected or plan_ready — both are acceptable terminal outcomes. - expect(['detected', 'plan_ready']).toContain(outcome); - // Fixture must not have been mutated regardless of outcome. - expect(fixtureStable).toBe(true); - } finally { - // Clean up fixture root. - try { fs.rmSync(path.dirname(fixture.workTree), { recursive: true, force: true }); } catch { /* ignore */ } - } - }, - PTY_LONG_MS, // 20 min wall clock - ); -}); diff --git a/test/skill-e2e-spec-execute.test.ts b/test/skill-e2e-spec-execute.test.ts deleted file mode 100644 index 787b91c72..000000000 --- a/test/skill-e2e-spec-execute.test.ts +++ /dev/null @@ -1,34 +0,0 @@ -/** - * /spec --execute end-to-end (periodic, paid, real-PTY). - * - * Asserts: when /spec --execute runs against a fixture prompt, it: - * 1. Refuses to draft on turn 1 (Phase 1 hard gate) - * 2. Reads code in Phase 3 (cites a real file path from the fixture repo) - * 3. Passes the quality gate (score >= 7) on a well-formed fixture - * 4. Spawns a fresh worktree on branch spec/- - * 5. Issues a final-confirm AskUserQuestion before the spawn - * - * Cost: ~$3-5/run, 5-8 min wall clock. Periodic — runs weekly via cron or - * on demand via `EVALS=1 EVALS_TIER=periodic bun run test:e2e`. - * - * TODO (v1.1): expand to test all 5 expansion paths and the plan-mode-aware - * Phase 5 branching (active vs inactive). Current implementation is the - * minimum smoke that proves --execute end-to-end works. - */ - -import { test } from 'bun:test'; -import { describeE2ETier } from './helpers/e2e-gate'; - -const describeE2E = describeE2ETier('periodic'); - -describeE2E('/spec --execute end-to-end (periodic)', () => { - // test.todo, not expect(true): the placeholder reported PASS on every - // periodic run while asserting nothing — a lying green with a 600s budget. - // The file itself stays: it is the periodic-tier surface registered in - // E2E_TIERS so the diff-based selector runs it when spec/ changes, and - // the deterministic template-invariant coverage in - // spec-template-invariants.test.ts + spec-template-sync.test.ts gates the - // gate tier. Implementation spec for the real PTY-driven test lives in - // the header TODO ("/spec --execute E2E full pipeline test (v1.1)"). - test.todo('phase gating + magical Phase 3 + quality gate + spawn — full pipeline'); -}); diff --git a/test/skill-llm-eval-spec.test.ts b/test/skill-llm-eval-spec.test.ts deleted file mode 100644 index 87922f365..000000000 --- a/test/skill-llm-eval-spec.test.ts +++ /dev/null @@ -1,35 +0,0 @@ -/** - * /spec LLM-judge eval (periodic, paid). - * - * Asserts: when /spec runs against a fixture vague request, the agent - * produces a spec body that scores >= 8/10 against an LLM judge using - * the contributor's 14 Quality Standards as the rubric. - * - * Cost: ~$0.15/run. Periodic — runs weekly via cron or on demand via - * `EVALS=1 EVALS_TIER=periodic bun run test:evals`. - * - * TODO (v1.1): expand fixture set to cover bug / feature / refactor / audit - * framings + project-level prompts (no concrete file mapping, exercises the - * Phase 3 fallback path). - */ - -import { describe, test } from 'bun:test'; - -const evalsEnabled = !!process.env.EVALS; -const describeEval = evalsEnabled ? describe : describe.skip; - -describeEval('/spec LLM-judge eval (periodic)', () => { - // test.todo, not expect(true): the placeholder reported PASS on every - // run while asserting nothing — a lying green with a 300s budget. The - // file stays as the periodic-tier selector surface for spec/ changes. - // - // Expected v1.1 implementation: - // 1. Pick fixture prompt from test/fixtures/spec/vague-bug.md - // 2. Spawn `claude -p` with /spec loaded, send the prompt + role-play - // five Phase 1 answers (from test/fixtures/spec/vague-bug-answers.json) - // 3. Capture final spec body - // 4. Dispatch to Claude judge with prompt encoding the 14 Quality - // Standards from spec/SKILL.md.tmpl - // 5. Assert numeric score >= 8 - test.todo('spec body scores >= 8/10 against 14-standard rubric on fixture request'); -}); diff --git a/test/skill-llm-eval.test.ts b/test/skill-llm-eval.test.ts index 3733f526d..d500b1f4b 100644 --- a/test/skill-llm-eval.test.ts +++ b/test/skill-llm-eval.test.ts @@ -12,7 +12,6 @@ import { afterAll, expect } from 'bun:test'; import { JUDGE_MS } from './helpers/eval-budgets'; -import Anthropic from '@anthropic-ai/sdk'; import * as fs from 'fs'; import * as path from 'path'; import { callJudge, judge, JudgeRefusalError, DEFAULT_JUDGE_MAX_TOKENS } from './helpers/llm-judge'; @@ -62,12 +61,11 @@ function readBrowseCommandSection(): string { } /** Slice a section out of the command-list section file, guarded non-empty. */ -function sliceBrowseSection(startHeader: string, endHeader?: string): string { +function sliceBrowseSection(startHeader: string): string { const content = readBrowseCommandSection(); const start = content.indexOf(startHeader); if (start < 0) throw new Error(`browse/sections/command-list.md: "${startHeader}" not found`); - const end = endHeader ? content.indexOf(endHeader) : -1; - const section = end > start ? content.slice(start, end) : content.slice(start); + const section = content.slice(start); if (section.trim().length < 200) { throw new Error(`browse/sections/command-list.md slice at "${startHeader}" is empty/stub — regenerate with: bun run gen:skill-docs`); } @@ -91,85 +89,45 @@ function testIfSelected(testName: string, fn: () => Promise, timeout: numb } describeIfSelected('LLM-as-judge quality evals', [ - 'command reference table', 'snapshot flags reference', - 'browse/SKILL.md reference', 'setup block', 'regression vs baseline', + 'browse/SKILL.md reference', 'setup block', ], () => { - testIfSelected('command reference table', async () => { - const t0 = Date.now(); - // Browse carve: the command reference lives in the generated on-demand - // section browse/sections/command-list.md now (read via non-empty guard). - const section = sliceBrowseSection('## Full Command List'); - - const scores = await judge('command reference table', section); - console.log('Command reference scores:', JSON.stringify(scores, null, 2)); - - // Completeness threshold is 3 (not 4) — the command reference table is - // intentionally terse (quick-reference format). The judge consistently scores - // completeness=3 because detailed argument docs live in per-command sections. - evalCollector?.addTest({ - name: 'command reference table', - suite: 'LLM-as-judge quality evals', - tier: 'llm-judge', - passed: scores.clarity >= 3 && scores.completeness >= 3 && scores.actionability >= 4, - duration_ms: Date.now() - t0, - cost_usd: 0.02, - judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning, - }); - - expect(scores.clarity).toBeGreaterThanOrEqual(3); - expect(scores.completeness).toBeGreaterThanOrEqual(3); - expect(scores.actionability).toBeGreaterThanOrEqual(4); - }, JUDGE_MS); - - testIfSelected('snapshot flags reference', async () => { - const t0 = Date.now(); - // Browse carve: snapshot flags live in browse/sections/command-list.md now, - // ordered before '## Full Command List' (the '## CSS Inspector' end boundary - // stayed in the skeleton). - const section = sliceBrowseSection('## Snapshot Flags', '## Full Command List'); - - const scores = await judge('snapshot flags reference', section); - console.log('Snapshot flags scores:', JSON.stringify(scores, null, 2)); - - evalCollector?.addTest({ - name: 'snapshot flags reference', - suite: 'LLM-as-judge quality evals', - tier: 'llm-judge', - passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4, - duration_ms: Date.now() - t0, - cost_usd: 0.02, - judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning, - }); - - expect(scores.clarity).toBeGreaterThanOrEqual(3); - expect(scores.completeness).toBeGreaterThanOrEqual(4); - expect(scores.actionability).toBeGreaterThanOrEqual(4); - }, JUDGE_MS); - testIfSelected('browse/SKILL.md reference', async () => { const t0 = Date.now(); - // Browse carve: flags + commands are the whole generated section file. + // Browse carve: snapshot flags + the full command list are the whole + // generated section file; one judge grades the union. Scores are also + // pinned against test/fixtures/eval-baselines.json (UPDATE_BASELINES=1 + // rewrites the pin). const section = sliceBrowseSection('## Snapshot Flags'); const scores = await judge('browse skill reference (flags + commands)', section); console.log('Browse SKILL.md scores:', JSON.stringify(scores, null, 2)); + const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json'); + const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8')); + const regressions = (['clarity', 'completeness', 'actionability'] as const) + .filter(dim => scores[dim] < baselines.browse_skill[dim]) + .map(dim => `browse_skill.${dim}: ${scores[dim]} < baseline ${baselines.browse_skill[dim]}`); + if (process.env.UPDATE_BASELINES) { + baselines.browse_skill = { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }; + fs.writeFileSync(baselinesPath, JSON.stringify(baselines, null, 2) + '\n'); + console.log('Updated eval baselines'); + } + evalCollector?.addTest({ name: 'browse/SKILL.md reference', suite: 'LLM-as-judge quality evals', tier: 'llm-judge', - passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4, + passed: scores.clarity >= 3 && scores.completeness >= 4 && scores.actionability >= 4 && regressions.length === 0, duration_ms: Date.now() - t0, cost_usd: 0.02, judge_scores: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability }, - judge_reasoning: scores.reasoning, + judge_reasoning: regressions.length ? `${scores.reasoning} | ${regressions.join('; ')}` : scores.reasoning, }); expect(scores.clarity).toBeGreaterThanOrEqual(3); expect(scores.completeness).toBeGreaterThanOrEqual(4); expect(scores.actionability).toBeGreaterThanOrEqual(4); + expect(regressions).toEqual([]); }, JUDGE_MS); testIfSelected('setup block', async () => { @@ -205,89 +163,6 @@ describeIfSelected('LLM-as-judge quality evals', [ expect(scores.clarity).toBeGreaterThanOrEqual(3); }, JUDGE_MS); - testIfSelected('regression vs baseline', async () => { - const t0 = Date.now(); - // Browse carve: the command reference lives in browse/sections/command-list.md. - const genSection = sliceBrowseSection('## Full Command List'); - - const baseline = `## Command Reference - -### Navigation -| Command | Description | -|---------|-------------| -| \`goto \` | Navigate to URL | -| \`back\` / \`forward\` | History navigation | -| \`reload\` | Reload page | -| \`url\` | Print current URL | - -### Interaction -| Command | Description | -|---------|-------------| -| \`click \` | Click element | -| \`fill \` | Fill input | -| \`select \` | Select dropdown | -| \`hover \` | Hover element | -| \`type \` | Type into focused element | -| \`press \` | Press key (Enter, Tab, Escape) | -| \`scroll [sel]\` | Scroll element into view | -| \`wait \` | Wait for element (max 10s) | -| \`wait --networkidle\` | Wait for network to be idle | -| \`wait --load\` | Wait for page load event | - -### Inspection -| Command | Description | -|---------|-------------| -| \`js \` | Run JavaScript | -| \`css \` | Computed CSS | -| \`attrs \` | Element attributes | -| \`is \` | State check (visible/hidden/enabled/disabled/checked/editable/focused) | -| \`console [--clear\\|--errors]\` | Console messages (--errors filters to error/warning) |`; - - const client = new Anthropic(); - const response = await client.messages.create({ - model: 'claude-sonnet-4-6', - max_tokens: 1024, - messages: [{ - role: 'user', - content: `You are comparing two versions of CLI documentation for an AI coding agent. - -VERSION A (baseline — hand-maintained): -${baseline} - -VERSION B (auto-generated from source): -${genSection} - -Which version is better for an AI agent trying to use these commands? Consider: -- Completeness (more commands documented? all args shown?) -- Clarity (descriptions helpful?) -- Coverage (missing commands in either version?) - -Respond with ONLY valid JSON: -{"winner": "A" or "B" or "tie", "reasoning": "brief explanation", "a_score": N, "b_score": N} - -Scores are 1-5 overall quality.`, - }], - }); - - const text = response.content[0].type === 'text' ? response.content[0].text : ''; - const jsonMatch = text.match(/\{[\s\S]*\}/); - if (!jsonMatch) throw new Error(`Judge returned non-JSON: ${text.slice(0, 200)}`); - const result = JSON.parse(jsonMatch[0]); - console.log('Regression comparison:', JSON.stringify(result, null, 2)); - - evalCollector?.addTest({ - name: 'regression vs baseline', - suite: 'LLM-as-judge quality evals', - tier: 'llm-judge', - passed: result.b_score >= result.a_score, - duration_ms: Date.now() - t0, - cost_usd: 0.02, - judge_scores: { a_score: result.a_score, b_score: result.b_score }, - judge_reasoning: result.reasoning, - }); - - expect(result.b_score).toBeGreaterThanOrEqual(result.a_score); - }, JUDGE_MS); }); // --- Part 7: QA skill quality evals (C6) --- @@ -523,59 +398,6 @@ score (1-5): 5 = perfectly consistent, 1 = contradictory`); }, JUDGE_MS); }); -// --- Part 7: Baseline score pinning (C9) --- - -describeIfSelected('Baseline score pinning', ['baseline score pinning'], () => { - const baselinesPath = path.join(ROOT, 'test', 'fixtures', 'eval-baselines.json'); - - testIfSelected('baseline score pinning', async () => { - const t0 = Date.now(); - if (!fs.existsSync(baselinesPath)) { - console.log('No baseline file found — skipping pinning check'); - return; - } - - const baselines = JSON.parse(fs.readFileSync(baselinesPath, 'utf-8')); - const regressions: string[] = []; - - // Browse carve: the command reference lives in browse/sections/command-list.md. - const cmdSection = sliceBrowseSection('## Full Command List'); - const cmdScores = await judge('command reference table', cmdSection); - - for (const dim of ['clarity', 'completeness', 'actionability'] as const) { - if (cmdScores[dim] < baselines.command_reference[dim]) { - regressions.push(`command_reference.${dim}: ${cmdScores[dim]} < baseline ${baselines.command_reference[dim]}`); - } - } - - if (process.env.UPDATE_BASELINES) { - baselines.command_reference = { - clarity: cmdScores.clarity, - completeness: cmdScores.completeness, - actionability: cmdScores.actionability, - }; - fs.writeFileSync(baselinesPath, JSON.stringify(baselines, null, 2) + '\n'); - console.log('Updated eval baselines'); - } - - const passed = regressions.length === 0; - evalCollector?.addTest({ - name: 'baseline score pinning', - suite: 'Baseline score pinning', - tier: 'llm-judge', - passed, - duration_ms: Date.now() - t0, - cost_usd: 0.02, - judge_scores: { clarity: cmdScores.clarity, completeness: cmdScores.completeness, actionability: cmdScores.actionability }, - judge_reasoning: passed ? 'All scores at or above baseline' : regressions.join('; '), - }); - - if (!passed) { - throw new Error(`Score regressions detected:\n${regressions.join('\n')}`); - } - }, JUDGE_MS); -}); - // --- Workflow SKILL.md quality evals (10 new tests for 100% coverage) --- /** diff --git a/test/skill-routing-e2e.test.ts b/test/skill-routing-e2e.test.ts index 438af6189..975990a62 100644 --- a/test/skill-routing-e2e.test.ts +++ b/test/skill-routing-e2e.test.ts @@ -590,6 +590,52 @@ export default app; } }, CAPTURE_MS); + testIfSelected('journey-negatives', async () => { + // Casual or off-topic prompts that share routing keywords ("wtf", + // "algorithm", "send it to the team") must not invoke a skill. Folded + // from the retired Opus 4.7 routing eval; same bound: at most one of the + // three may route. + const cases = [ + { name: 'neg-syntax-q', prompt: 'wtf does this Python list comprehension syntax even mean, [x for x in y if z]?' }, + { name: 'neg-algo-q', prompt: 'does this bubble sort algorithm actually work in O(n log n)?' }, + { name: 'neg-slack-send', prompt: 'can you help me write the slack message? I want to send it to the team.' }, + ]; + const tmpDir = createRoutingWorkDir('negatives'); + try { + const results = await Promise.all(cases.map(async c => { + const result = await runSkillTest({ + prompt: c.prompt, + workingDirectory: tmpDir, + maxTurns: 2, + allowedTools: ['Skill', 'Read'], + timeout: JUDGE_MS, + testName: `journey-negatives-${c.name}`, + runId, + }); + const skillCalls = result.toolCalls.filter(tc => tc.tool === 'Skill'); + const actualSkill = skillCalls.length > 0 ? skillCalls[0]?.input?.skill : undefined; + logCost(`journey: journey-negatives ${c.name}`, result); + evalCollector?.addTest({ + name: `journey-negatives-${c.name}`, + suite: 'Skill Routing E2E', + tier: 'e2e', + passed: actualSkill === undefined, + duration_ms: result.duration, + cost_usd: result.costEstimate.estimatedCost, + transcript: result.transcript, + output: `routed=${actualSkill ?? '(none)'}`, + turns_used: result.costEstimate.turnsUsed, + exit_reason: result.exitReason, + }); + return { name: c.name, actualSkill }; + })); + const routed = results.filter(r => r.actualSkill !== undefined); + expect(routed.length, `negatives routed: ${routed.map(r => `${r.name}→${r.actualSkill}`).join(', ')}`).toBeLessThanOrEqual(1); + } finally { + fs.rmSync(tmpDir, { recursive: true, force: true }); + } + }, CAPTURE_MS); + testIfSelected('journey-visual-qa', async () => { const tmpDir = createRoutingWorkDir('visual-qa'); try { diff --git a/test/test-free-shards.test.ts b/test/test-free-shards.test.ts index c14d9faf2..bb50e6bf3 100644 --- a/test/test-free-shards.test.ts +++ b/test/test-free-shards.test.ts @@ -181,7 +181,6 @@ describe('test-free-shards: enumeration', () => { expect(isFreeTestFile('test/skill-llm-eval.test.ts')).toBe(false); expect(isFreeTestFile('test/codex-e2e.test.ts')).toBe(false); expect(isFreeTestFile('test/codex-e2e-sol-scope.test.ts')).toBe(false); - expect(isFreeTestFile('test/gemini-e2e.test.ts')).toBe(false); }); test('collectFreeTestFiles returns sorted, deduped, only-free list', () => { diff --git a/test/touchfiles.test.ts b/test/touchfiles.test.ts index 0f964517b..81698fff6 100644 --- a/test/touchfiles.test.ts +++ b/test/touchfiles.test.ts @@ -83,8 +83,7 @@ describe('selectTests', () => { // These cases use an outside-only/Code Quality excerpt or descriptive metadata. .filter(id => !['outside-plan-disabled-no-fallback', 'plan-ceo-review-prosons-cadence', 'plan-review-prosons-format', 'shared-libs-plan-callers'].includes(id)); - const existing = ['learnings-show', 'codex-plan-ceo-format-mode', 'codex-plan-ceo-format-approach', - 'codex-plan-eng-format-coverage', 'codex-plan-eng-format-kind']; + const existing = ['learnings-show']; const result = selectTests(['scripts/resolvers/learnings.ts'], E2E_TOUCHFILES); expect(result.reason).toBe('diff'); expect(result.selected.sort()).toEqual([...new Set([...consumers, ...existing])].sort()); @@ -161,12 +160,8 @@ describe('selectTests', () => { expect(fs.readFileSync(path.join(ROOT, `${output}.tmpl`), 'utf8')).toContain(`{{${token}}}`); } const generated = selectTests(consumers.map(([output]) => output), E2E_TOUCHFILES); - // These two CEO-format cases already depend on every resolver through - // scripts/resolvers/**; keep that existing selection alongside consumers. // The bounded Code Quality fixture stops before Test review. - const expected = [...new Set([...generated.selected.filter(id => id !== 'shared-libs-plan-callers' && !SHIP_GUARD_ONLY.includes(id)), - 'codex-plan-ceo-format-mode', 'codex-plan-ceo-format-approach', - ])].sort(); + const expected = [...new Set(generated.selected.filter(id => id !== 'shared-libs-plan-callers' && !SHIP_GUARD_ONLY.includes(id)))].sort(); const actual = selectTests(['scripts/resolvers/testing.ts'], E2E_TOUCHFILES); expect(actual.reason).toBe('diff'); expect(actual.selected.sort()).toEqual(expected); @@ -373,11 +368,9 @@ describe('selectTests', () => { expect(result.selected).toContain('plan-ceo-split-overflow'); // v2 plan Phase B carve: the section-loading E2E depends on plan-ceo-review/**. expect(result.selected).toContain('plan-ceo-section-loading'); - expect(result.selected).toContain('codex-plan-ceo-format-mode'); - expect(result.selected).toContain('codex-plan-ceo-format-approach'); expect(result.selected).toContain('outside-plan-disabled-no-fallback'); - expect(result.selected.length).toBe(25); - expect(result.skipped.length).toBe(Object.keys(E2E_TOUCHFILES).length - 25); + expect(result.selected.length).toBe(23); + expect(result.skipped.length).toBe(Object.keys(E2E_TOUCHFILES).length - 23); }); test('global touchfile triggers ALL tests', () => { @@ -398,7 +391,6 @@ describe('selectTests', () => { expect(result.reason).toBe('diff'); expect(result.selected).toContain('autoplan-chain-pty'); expect(result.selected).toContain('plan-ceo-mode-routing'); - expect(result.selected).not.toContain('codex-plan-ceo-format-mode'); expect(result.selected).not.toContain('retro'); }); @@ -604,7 +596,7 @@ describe('TOUCHFILES completeness', () => { ); const unique = registeredJudgeTestNames(llmContent); - expect(unique).toHaveLength(27); + expect(unique).toHaveLength(23); const missing = unique.filter(name => !(name in LLM_JUDGE_TOUCHFILES)); if (missing.length > 0) { @@ -623,7 +615,7 @@ describe('TOUCHFILES completeness', () => { testIfSelected('unmapped judge case', async () => {}, 120_000); `; const names = registeredJudgeTestNames(withUnmappedCase); - expect(names).toHaveLength(28); + expect(names).toHaveLength(24); expect(names.filter(name => !(name in LLM_JUDGE_TOUCHFILES))).toEqual(['unmapped judge case']); }); diff --git a/test/workflow-judge-cache.test.ts b/test/workflow-judge-cache.test.ts index 5855f2d0b..4c33079b3 100644 --- a/test/workflow-judge-cache.test.ts +++ b/test/workflow-judge-cache.test.ts @@ -163,7 +163,7 @@ test('workflow registration preserves model work and reserves only terminal-reco expect(source).toContain('const WORKFLOW_JUDGE_RECORD_MS = 5_000;'); expect(source).toContain('const WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000;'); expect(source.match(/\}, WORKFLOW_JUDGE_TEST_MS\);/g)).toHaveLength(16); - expect(source.match(/\}, JUDGE_MS\);/g)).toHaveLength(11); + expect(source.match(/\}, JUDGE_MS\);/g)).toHaveLength(7); }); function actualCallback(f: ReturnType, overrides: {