/** Test value bar behavior in /ship Step 7, the /review testing specialist and * /test-audit. Each case is one bounded capture on a small fixture: * ship-coverage-value a ★-only path lands in weak_gaps and X <= Y * review-test-value a source-grep test and a test-only export are * INFORMATIONAL findings; the SKILL.md golden is not * test-audit-report-only the same fixture yields complete retirement cards, * keeps the golden and edits nothing */ import { afterAll } from 'bun:test'; import * as fs from 'node:fs'; import * as path from 'node:path'; import * as os from 'node:os'; import { spawnSync } from 'node:child_process'; import { CAPTURE_MS } from './helpers/eval-budgets'; import { runSkillTest } from './helpers/session-runner'; import { ROOT, runId, describeIfSelected, testIfSelected, copyDirSync, logCost, createEvalCollector, finalizeEvalCollector } from './helpers/e2e-helpers'; import { extractSkillBody } from './helpers/skill-fixture'; import { createCoverageAuditFixture } from './fixtures/coverage-audit-fixture'; import { createTestValueFixture, lastJsonLine, jsonFindings, LOW_VALUE_TESTS } from './helpers/test-value-fixture'; import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt'; import { resolveEvalModel } from '../lib/eval-model'; import { RETIREMENT_FIELDS } from '../scripts/resolvers/test-value'; import type { SkillTestResult } from './helpers/session-runner'; const evalCollector = createEvalCollector('e2e'); const RUNNER_MS = CAPTURE_MS - 3 * OFFICE_HOURS_BUN_GRACE_MS; function trackedChanges(cwd: string): string { return spawnSync('git', ['status', '--porcelain', '--untracked-files=no'], { cwd, encoding: 'utf8', timeout: 10_000 }).stdout.trim(); } const SMOKE_TEST = ` import { refundPayment } from '../src/billing'; test('refundPayment does not throw', () => { expect(() => refundPayment('pay_1', 'duplicate')).not.toThrow(); }); `; type Case = { id: string; skill: string; suite: string; setup: (cwd: string) => void; prompt: (cwd: string, reportDir: string) => string; validate: (result: SkillTestResult, cwd: string, reportDir: string) => void; }; const CASES: Case[] = [ { id: 'ship-coverage-value', skill: 'ship', suite: 'Ship Test Value E2E', setup: cwd => { createCoverageAuditFixture(cwd); fs.appendFileSync(path.join(cwd, 'test/billing.test.ts'), SMOKE_TEST); }, prompt: cwd => `Read ship/SKILL.md and ship/sections/test-coverage.md for the current ship workflow. You are on the feature/billing branch. The base branch is main. There is no remote. Run ONLY the Step 7 coverage audit, inline, as the audit subagent would, applying it directly to ${cwd}/src/billing.ts and its tests in ${cwd}/test/billing.test.ts (a targeted audit with no branch diff). Generation: audit-only; passes used: 0 of 2. Do not dispatch subagents, write tests or modify files. Skip every other step. End with the audit's LAST-line JSON exactly as Step 7 specifies.`, validate: result => { const json = lastJsonLine(result.output || ''); if (!json) throw new Error('ship coverage: no last-line JSON'); if (typeof json.coverage_pct !== 'number' || typeof json.coverage_pct_value !== 'number') throw new Error(`ship coverage: numeric coverage_pct and coverage_pct_value required, got ${JSON.stringify(json)}`); if (json.coverage_pct < json.coverage_pct_value) throw new Error(`ship coverage: coverage_pct ${json.coverage_pct} < coverage_pct_value ${json.coverage_pct_value}`); const weak = Array.isArray(json.weak_gaps) ? json.weak_gaps : []; if (!weak.some((gap: any) => /refund/i.test(JSON.stringify(gap)))) throw new Error(`ship coverage: the ★-only refundPayment path must be in weak_gaps, got ${JSON.stringify(json.weak_gaps)}`); }, }, { id: 'review-test-value', skill: 'review', suite: 'Review Test Value E2E', setup: cwd => createTestValueFixture(cwd), prompt: () => `Read review/SKILL.md and review/sections/review-army.md for the current review workflow. Apply ONLY the testing specialist checklist in review/specialists/testing.md to this branch's diff (\`git diff main...HEAD\`), as the testing specialist would. Do not dispatch other specialists, fix anything or modify files. Output the specialist's JSON findings, one per line.`, validate: (result, cwd) => { const findings = jsonFindings(result.output || ''); const about = (needle: RegExp) => findings.filter(finding => needle.test(`${finding.path} ${finding.summary}`)); const grep = about(/pricing-source/); const seam = about(new RegExp(`pricing-reset|${LOW_VALUE_TESTS.testOnlySymbol}`)); if (!grep.length || !grep.every(finding => finding.severity === 'INFORMATIONAL')) throw new Error(`review: source-grep test needs an INFORMATIONAL finding, got ${JSON.stringify(grep)}`); if (!seam.length || !seam.every(finding => finding.severity === 'INFORMATIONAL')) throw new Error(`review: test-only export needs an INFORMATIONAL finding, got ${JSON.stringify(seam)}`); if (!seam.some(finding => /non_test_callers/.test(JSON.stringify(finding.evidence ?? '')) && /git grep/.test(JSON.stringify(finding.evidence ?? '')))) throw new Error('review: test-only export evidence must record non_test_callers and the git grep search'); const golden = about(/skill-golden/); if (golden.length) throw new Error(`review: the SKILL.md golden test must not be flagged, got ${JSON.stringify(golden)}`); if (trackedChanges(cwd)) throw new Error('review: tracked files changed'); }, }, { id: 'test-audit-report-only', skill: 'test-audit', suite: 'Test Audit Report-Only E2E', setup: cwd => createTestValueFixture(cwd), prompt: (_cwd, reportDir) => `Read test-audit/SKILL.md and run /test-audit on this repository, report-only. Treat this as a headless session: ask no questions, approve no batch, edit no file in the repository. There is no gstack install here, so skip the SLUG setup line and write the report to ${reportDir}/test-audit.md and its JSON sidecar to ${reportDir}/test-audit.json instead. Stop after Step 4.`, validate: (_result, cwd, reportDir) => { const sidecarPath = path.join(reportDir, 'test-audit.json'); if (!fs.existsSync(path.join(reportDir, 'test-audit.md'))) throw new Error('test-audit: report missing'); if (!fs.existsSync(sidecarPath)) throw new Error('test-audit: JSON sidecar missing'); const sidecar = JSON.parse(fs.readFileSync(sidecarPath, 'utf8')); const candidates: any[] = Array.isArray(sidecar.candidates) ? sidecar.candidates : []; const retiring = candidates.filter(candidate => candidate.verdict !== 'retain'); for (const needle of [/pricing-source/, new RegExp(`pricing-reset|${LOW_VALUE_TESTS.testOnlySymbol}`)]) { const match = retiring.find(candidate => needle.test(String(candidate.test))); if (!match) throw new Error(`test-audit: missing candidate ${needle}, got ${JSON.stringify(candidates.map(c => c.test))}`); const missing = RETIREMENT_FIELDS.filter(field => !String(match.retirement_card?.[field] ?? '').trim()); if (missing.length) throw new Error(`test-audit: ${match.test} retirement card missing ${missing.join(', ')}`); } if (retiring.some(candidate => /skill-golden/.test(String(candidate.test)))) throw new Error('test-audit: the SKILL.md golden must be retained'); if (trackedChanges(cwd)) throw new Error('test-audit: tracked files changed'); }, }, ]; for (const entry of CASES) describeIfSelected(entry.suite, [entry.id], () => { testIfSelected(entry.id, async () => { let cwd: string | undefined; let reportDir: string | undefined; try { await runRecordedOfficeHoursAttempt({ collector: evalCollector, name: entry.id, suite: entry.suite, model: process.env.EVALS_MODEL ?? resolveEvalModel('capture'), budgetMs: CAPTURE_MS - OFFICE_HOURS_BUN_GRACE_MS, run: async signal => { cwd = fs.mkdtempSync(path.join(os.tmpdir(), `skill-e2e-${entry.id}-`)); reportDir = fs.mkdtempSync(path.join(os.tmpdir(), `skill-e2e-${entry.id}-report-`)); entry.setup(cwd); copyDirSync(path.join(ROOT, entry.skill), path.join(cwd, entry.skill)); fs.writeFileSync(path.join(cwd, entry.skill, 'SKILL.md'), extractSkillBody(path.join(ROOT, entry.skill))); fs.writeFileSync(path.join(cwd, '.git', 'info', 'exclude'), `${entry.skill}/\n`); return runSkillTest({ prompt: entry.prompt(cwd, reportDir), workingDirectory: cwd, maxTurns: 25, allowedTools: ['Bash', 'Read', 'Write', 'Glob', 'Grep'], timeout: RUNNER_MS, testName: entry.id, runId, signal, }); }, validate: result => { logCost(entry.id, result); if (result.exitReason !== 'success') throw new Error(`${entry.id}: ${result.exitReason}`); entry.validate(result, cwd!, reportDir!); }, }); } finally { for (const dir of [cwd, reportDir]) if (dir) try { fs.rmSync(dir, { recursive: true, force: true }); } catch {} } }, CAPTURE_MS); }); afterAll(async () => { await finalizeEvalCollector(evalCollector); });