v1.91.9.0 feat: test value bar in plan-eng-review, review, qa and ship, plus /test-audit (#2998)

This commit is contained in:
Garry Tan authored and GitHub committed 2026-09-29 14:35:00 -07:00
1 parent 943105f109
commit 96764e80a6
56 files changed
+2445 -216

No files matched your search

+3
View File
@@ -39,6 +39,7 @@ import { generateAsideSetup, generateAsideCookbook, generateAsideResearch, gener
import { generateCommandReference, generateSnapshotFlags, generateBrowseSetup, generateBrowseFallback } from './browse';
import { generateDesignDocDiscovery } from './design-doc-discovery';
import { generateSharedLibsRubric } from './shared-libs';
import { generateTestValueBar, generateTestValueMessage } from './test-value';
import { generateQAScope, generateQAExploratory, generateQAFunctional, generateQAResource, generateQAReview, generateQAReviewPreflight, generateQAMethodReads } from './qa';
export const RESOLVERS: Record<string, ResolverFn> = {
@@ -100,6 +101,8 @@ export const RESOLVERS: Record<string, ResolverFn> = {
TEST_COVERAGE_AUDIT_PLAN: generateTestCoverageAuditPlan,
TEST_COVERAGE_AUDIT_SHIP: generateTestCoverageAuditShip,
TEST_COVERAGE_GATE_SHIP: generateTestCoverageGateShip,
TEST_VALUE_BAR: generateTestValueBar,
TEST_VALUE_MESSAGE: generateTestValueMessage,
TEST_FAILURE_TRIAGE: generateTestFailureTriage,
SPEC_REVIEW_LOOP: generateSpecReviewLoop,
DESIGN_SKETCH: generateDesignSketch,
+171
View File
@@ -0,0 +1,171 @@
import type { TemplateContext } from './types';
// Test value bar shared by plan-eng-review and ship (through the coverage audit),
// qa/qa-only ({{TEST_VALUE_BAR:qa}}) and test-audit ({{TEST_VALUE_BAR:audit}}).
// The static review/specialists/testing.md repeats these constants and is kept
// in sync by test/test-value-bar.test.ts.
// Adapted from openclaw/openclaw@a214e76, .agents/skills/test-audit/SKILL.md.
export const TEST_VALUE_BAR_MODES = ['plan', 'ship', 'qa', 'audit'] as const;
export type TestValueBarMode = (typeof TEST_VALUE_BAR_MODES)[number];
export const QUESTIONS = [
'What observable behavior, invariant or independent contract does it protect?',
'What credible regression makes it fail?',
'Why does existing coverage not already catch that? Prefer adding a row to an existing table-driven test or shared fixture over a near-duplicate.',
'Does it need a production seam (export, flag, wrapper, injection hook) that no production caller needs? If yes, test at the real boundary instead.',
] as const;
export const VALUE_CARD_FIELDS = ['protects', 'fails_when', 'why_new', 'seam'] as const;
export const CARD_FIELD_MAX_BYTES = 160;
export const CATALOG = [
'assertion-free coverage probes',
'self-comparisons and identity copies',
'copied fixtures, inventories or export lists',
'exact source, import or string greps that are not a declared contract',
'private predicate or call-shape tests duplicated at a real boundary',
'duplicate invocations of the same contract',
"per-caller replays of a shared helper's tests",
'tests whose only purpose is keeping a test-only export, global or wrapper alive',
'production code whose only callers are tests',
] as const;
export const RETIREMENT_FIELDS = ['test', 'detects', 'non_test_callers', 'search_command', 'stronger_proof', 'history', 'unlocks', 'validation'] as const;
export const REVIEW_EVIDENCE_FIELDS = ['detects', 'non_test_callers', 'search_command', 'stronger_proof'] as const;
export const REASON_CODES = ['duplicate_protects', 'needs_seam', 'incomplete_card', 'no_credible_regression', 'covered_elsewhere', 'implementation_coupled'] as const;
export const WEAK_REASONS = ['star_one', 'gate_failed', 'unrated'] as const;
export const PRAGMA = 'gstack:test-value keep';
export const SWEEP_POINTER = 'Repo-wide sweep: run /test-audit.';
export const RETENTION_ONE_LINER = 'Retention bar: keep a test that independently enforces a public API, protocol, config, migration, storage, security, platform, default, prompt-byte, generated-output (golden), package, release or architecture contract; static or slow is no reason to delete.';
export const CALLER_SEARCH_COMMAND = "git grep -n -F -w -e '<symbol>' -- . ':!test/' ':!tests/' ':!spec/' ':!**/__tests__/**' ':!**/*.test.*' ':!**/*.spec.*' ':!**/*_test.*' ':!**/test_*.py'";
export const CALLER_SYMBOL_PATTERN = '^[A-Za-z_][A-Za-z0-9_]*$';
export const DOCS_PAGE = 'docs/test-value-bar.md';
export const MESSAGES = {
ratingUnavailable: {
message: 'rating unavailable: the read-only rating dispatch failed or timed out, so the coverage gate is skipped for this run. Re-run Step 7 to re-rate the tests.',
anchor: 'rating-unavailable',
},
valueCoverageUnavailable: {
message: 'value-weighted coverage unavailable (outdated installed skill); run /gstack-upgrade. The gate used coverage_pct (any test) this run.',
anchor: 'value-weighted-coverage-unavailable',
},
inconsistentCoverage: {
message: 'inconsistent coverage inputs: coverage_pct_value was above coverage_pct, so it was clamped to coverage_pct. Re-run Step 7 if the numbers look wrong.',
anchor: 'inconsistent-coverage-inputs',
},
malformedKey: {
message: 'malformed <key> ignored: the audit returned the wrong type, so it counts as empty. The likely cause is an outdated installed skill; run /gstack-upgrade.',
anchor: 'malformed-key-ignored',
},
allRejected: {
message: 'all <N> generated tests rejected by machine checks; see tests_rejected. The gate proceeds with the unchanged value-weighted coverage.',
anchor: 'all-generated-tests-rejected',
},
baseControlUnavailable: {
message: 'base control unavailable: <reason>. The fails-at-HEAD result still stands. To check by hand: `git worktree add --detach <tmp> <base>`; copy the test and its new fixtures to the same paths; run the detected test command in <tmp>; `git worktree remove --force <tmp>`. Then report `passes at base: manual`.',
anchor: 'base-control-unavailable',
},
callerCheckUnavailable: {
message: 'caller check unavailable: <command or "unsupported symbol">. The finding stays INFORMATIONAL and nothing is proposed for deletion; run the search by hand to complete the evidence.',
anchor: 'caller-check-unavailable',
},
unknownMode: {
message: 'Unknown TEST_VALUE_BAR mode <x>; expected plan|ship|qa|audit. Fix the placeholder or add the mode in scripts/resolvers/test-value.ts.',
anchor: 'unknown-test-value-bar-mode',
},
} as const;
export type MessageKey = keyof typeof MESSAGES;
export function degradedMessage(ctx: TemplateContext, key: MessageKey): string {
const { message, anchor } = MESSAGES[key];
return `${message} (see ${ctx.paths.skillRoot}/${DOCS_PAGE}#${anchor})`;
}
export function generateTestValueMessage(ctx: TemplateContext, args?: string[]): string {
const key = args?.[0] ?? '';
if (!(key in MESSAGES)) throw new Error(`Unknown TEST_VALUE_MESSAGE key ${key}; expected ${Object.keys(MESSAGES).join('|')} (scripts/resolvers/test-value.ts)`);
return degradedMessage(ctx, key as MessageKey);
}
// Measured renders plus 15%: plan 1897, ship 2742, qa 1036, audit 3712 bytes.
export const TEST_VALUE_BAR_MAX_BYTES: Record<TestValueBarMode, number> = { plan: 2182, ship: 3154, qa: 1192, audit: 4269 };
export function clampCardField(value: string): string {
const bytes = Buffer.from(value, 'utf8');
if (bytes.length <= CARD_FIELD_MAX_BYTES) return value;
let end = CARD_FIELD_MAX_BYTES - 3;
while (end > 0 && (bytes[end]! & 0xc0) === 0x80) end--;
return `${bytes.subarray(0, end).toString('utf8')}...`;
}
export function renderValueCard(card: Record<(typeof VALUE_CARD_FIELDS)[number], string>): string {
return `Value: ${VALUE_CARD_FIELDS.map(field => `${field}=${clampCardField(card[field])}`).join('; ')}`;
}
const EXAMPLE_CARD = renderValueCard({
protects: 'refundPayment rejects an empty reason',
fails_when: 'the reason guard is removed or inverted',
why_new: 'billing.test.ts covers processPayment only',
seam: 'none',
});
const EXAMPLE_REJECTED = 'Rejected (covered_elsewhere): "checkout renders"; checkout.e2e.ts:15 covers it, so extend that test.';
function questionList(mode: TestValueBarMode): string {
const numbered = QUESTIONS.map((question, index) => `${index + 1}. ${question}`);
return mode === 'qa' ? numbered.slice(2).join('\n') : numbered.join('\n');
}
function cardRules(mode: TestValueBarMode): string {
const where = {
plan: 'One card per Critical Path and Edge Case in the Test Plan Artifact.',
ship: 'Write it as a header comment in each generated test, next to the attribution (wrap, do not truncate); with no known comment syntax, put it in the PR body\'s Test value details.',
qa: 'Put it in the 8e.5 record (/qa) or under each proposed test (/qa-only).',
audit: 'Read cards from test header comments when present.',
}[mode];
return `Value card: \`Value: protects=<...>; fails_when=<...>; why_new=<...>; seam=none\` (seam: \`none\` or its name); each field at most ${CARD_FIELD_MAX_BYTES} UTF-8 bytes here (clamp to 157 plus \`...\`; JSON keeps full values). ${where} A missing upstream card never blocks: derive it; ignore unknown fields.
Example: ${EXAMPLE_CARD}
${EXAMPLE_REJECTED}`;
}
const STAR_RULE = 'Weak tests (★ smoke/existence/trivial, gate-failing or unrated) never count as coverage. X = paths with a ★★/★★★ test / total paths (value-weighted; the gate uses X); Y = paths with any test / total paths.';
const WEAK_PATH_RULE = `Total paths = the diff's codepath trace, max 30; zero skips the gate. A path with only weak tests is uncovered in X, covered in Y, and goes to \`weak_gaps\` (reason \`${WEAK_REASONS.join('|')}\`), not \`gaps\`. Rate stars only for tests reachable from changed paths.`;
const RED_PROOF = `Regression proof: a regression test must fail at HEAD before any repair, in its own assertion (a pass at HEAD drops the regression label; an import, fixture or env failure is a test defect: correct once or drop). It must pass at base as the control (an assertion failure there marks it invalid; any other failure is "base control unavailable: collection error") and pass after the repair. Record: \`Regression proof — fails at HEAD: yes · passes at base: yes | unavailable (<reason>) | manual · passes after fix: yes | pending\`.`;
function auditSections(): string {
return `Low-value catalog (a match fails the gate unless the retention bar names the contract it guards):
${CATALOG.map(entry => `- ${entry}`).join('\n')}
Retention bar: keep a test that independently enforces a public API, protocol, config, migration, storage, security, platform, default, prompt-byte, generated-output (SKILL.md golden), package, release or architecture contract; call order when order is observable; source inspection when it is the cheapest independent guard. Never retire anything reachable from the package entrypoint (\`package.json\` exports/main, index re-exports). Static or slow is not a reason to delete. Skip a test carrying \`${PRAGMA} reason="<why>"\` and list it as suppressed.
Retirement card, complete before any edit: ${RETIREMENT_FIELDS.map(field => `\`${field}\``).join(', ')}. Caller check for a symbol matching \`${CALLER_SYMBOL_PATTERN}\` (otherwise "caller check unavailable: unsupported symbol"): \`${CALLER_SEARCH_COMMAND}\`; record the command, exclusions and hit count. The evidence is grep-only (no re-exports, dynamic dispatch or generated code), so production code is retired only when the repo's typecheck/build or dead-code tool passes with it removed in a scratch worktree.`;
}
export function generateTestValueBar(_ctx: TemplateContext, args?: string[]): string {
const mode = args?.[0] as TestValueBarMode;
if (!TEST_VALUE_BAR_MODES.includes(mode)) throw new Error(MESSAGES.unknownMode.message.replace('<x>', String(args?.[0])));
const parts = [
`**Test value bar.** ${mode === 'qa' ? 'Before writing the test (the reproduced bug answers what it protects and what makes it fail):' : 'Propose or write a test only with all four answers; otherwise extend an existing test or drop it:'}`,
questionList(mode),
cardRules(mode),
];
if (mode !== 'qa') parts.splice(2, 0, 'A test that breaks under a behavior-preserving refactor asserts implementation: rewrite it at the owning boundary, unless exact output is the declared contract (goldens, prompt bytes, wire formats).');
if (mode === 'plan') parts.push(`${STAR_RULE} /ship computes them; here every proposed test needs a card.`, RETENTION_ONE_LINER);
if (mode === 'ship') parts.push(`${STAR_RULE} ${WEAK_PATH_RULE}`, RETENTION_ONE_LINER);
if (mode === 'ship' || mode === 'audit') parts.push(RED_PROOF);
if (mode === 'audit') parts.push(auditSections());
const rendered = parts.join('\n\n');
const bytes = Buffer.byteLength(rendered, 'utf8');
const budget = TEST_VALUE_BAR_MAX_BYTES[mode];
if (bytes > budget) throw new Error(`TEST_VALUE_BAR mode '${mode}' renders ${bytes} bytes, budget ${budget} (over by ${bytes - budget}). Trim the mode's section in scripts/resolvers/test-value.ts or raise the ceiling with a reason.`);
return rendered;
}
+98 -81
View File
@@ -1,5 +1,6 @@
import type { TemplateContext } from './types';
import { asideExecPrelude } from './aside';
import { generateTestValueBar, degradedMessage, REASON_CODES, SWEEP_POINTER, type TestValueBarMode } from './test-value';
export function generateTestBootstrap(ctx: TemplateContext): string {
return `## Test Framework Bootstrap
@@ -196,38 +197,59 @@ Only commit if there are changes. Stage all bootstrap files (config, test direct
// ─── Test Coverage Audit ────────────────────────────────────
//
// Shared methodology for codepath tracing, ASCII diagrams, and test gap analysis.
// Three modes, three placeholders, one inner function:
// Two modes, one inner function; both embed the test value bar (test-value.ts):
//
// {{TEST_COVERAGE_AUDIT_PLAN}} → plan-eng-review: adds missing tests to the plan
// {{TEST_COVERAGE_AUDIT_SHIP}} → ship: auto-generates tests, coverage summary
// {{TEST_COVERAGE_AUDIT_REVIEW}} → review: generates tests via Fix-First (ASK)
// {{TEST_COVERAGE_AUDIT_SHIP}} → ship: generates tests, coverage summary
// {{TEST_COVERAGE_GATE_SHIP}} → ship: parent-owned coverage gate
//
// ┌────────────────────────────────────────────────┐
// │ generateTestCoverageAuditInner(mode) │
// │ │
// │ SHARED: framework detect, codepath trace, │
// │ ASCII diagram, quality rubric, E2E matrix, │
// │ regression rule │
// │ │
// │ plan: edit plan file, write artifact │
// │ ship: auto-generate tests, write artifact │
// │ review: Fix-First ASK, INFORMATIONAL gaps │
// └────────────────────────────────────────────────┘
// /review uses its static testing specialist (review/specialists/testing.md).
type CoverageAuditMode = 'plan' | 'ship' | 'review';
export type CoverageAuditMode = Extract<TestValueBarMode, 'plan' | 'ship'>;
function generateTestCoverageAuditInner(mode: CoverageAuditMode, part: 'audit' | 'gate' = 'audit'): string {
function generateBaseControl(ctx: TemplateContext): string {
return `**Red-first proof.** Apply the value bar's Regression proof to every regression test: the diff at HEAD is the pre-fix code, so run the new test at HEAD before any repair. Then, unless the parent says \`Base control: off\`, run this base control once per regression test in diff order, within a 3-minute total per /ship run (\`Base control budget:\` seconds per test, default 90). Past the total, record "base control unavailable: budget" and report "N of M regression tests got base control". The block is one shell invocation; it installs nothing and runs no build or postinstall.
\`\`\`bash
# Set: BASE = the base branch this /ship run resolved; TEST = the test file; FIXTURES = new
# test-only fixtures it imports (repo-relative, may be empty); RUN = the detected
# runner for one file (e.g. "bun test $TEST"); BUDGET = seconds for this run (default 90).
ROOT=$(git rev-parse --show-toplevel)
CTL_TMP=$(mktemp -d "\${TMPDIR:-/tmp}/gstack-base-control.XXXXXX")
cleanup() { git -C "$ROOT" worktree remove --force "$CTL_TMP/wt" >/dev/null 2>&1; rm -rf "$CTL_TMP"; [ -e "$CTL_TMP" ] && echo "BASE_CONTROL_LEFTOVER: $CTL_TMP (run: git worktree prune)"; }
trap cleanup EXIT INT TERM
(
[ -f "$ROOT/package.json" ] || { echo "BASE_CONTROL: unavailable (ecosystem)"; exit 0; }
git -C "$ROOT" remote get-url origin >/dev/null 2>&1 || { echo "BASE_CONTROL: unavailable (no base remote)"; exit 0; }
timeout 30 git -C "$ROOT" fetch --quiet origin "$BASE" || { echo "BASE_CONTROL: unavailable (base not fetched)"; exit 0; }
git -C "$ROOT" worktree add --quiet --detach "$CTL_TMP/wt" "origin/$BASE" >/dev/null 2>&1 || { echo "BASE_CONTROL: unavailable (worktree add failed)"; exit 0; }
for f in $TEST $FIXTURES; do mkdir -p "$CTL_TMP/wt/$(dirname "$f")" && cp "$ROOT/$f" "$CTL_TMP/wt/$f"; done
[ -d "$ROOT/node_modules" ] && ln -s "$ROOT/node_modules" "$CTL_TMP/wt/node_modules"
cd "$CTL_TMP/wt" && timeout "\${BUDGET:-90}" sh -c "$RUN" > "$CTL_TMP/out" 2>&1; rc=$?
tail -n 40 "$CTL_TMP/out"
if [ "$rc" -eq 0 ]; then echo "BASE_CONTROL: passes at base"
elif [ "$rc" -eq 124 ]; then echo "BASE_CONTROL: unavailable (budget)"
else echo "BASE_CONTROL: fails at base (exit $rc)"; fi
)
\`\`\`
Classify "fails at base" from the output: a failure in the test's own assertion marks it invalid (correct once or drop it); an import, collection, missing generated artifact or dependency failure is "base control unavailable: collection error" and the test stays. Print each unavailable result as: ${degradedMessage(ctx, 'baseControlUnavailable')}
Return \`"regression_proof":{"red_at_head":N,"base_green":N,"base_unavailable":N}\` counts in the JSON and each test's record line in the diagram.`;
}
const COVERAGE_GOAL = 'Coverage goal: every changed behavior is protected by a test that would catch a real regression. Test count is not a goal.';
export function generateTestCoverageAuditInner(ctx: TemplateContext, mode: CoverageAuditMode, part: 'audit' | 'gate' = 'audit'): string {
const sections: string[] = [];
const subheading = mode === 'plan' ? '####' : '###';
let gate = '';
// ── Intro (mode-specific) ──
if (mode === 'ship') {
sections.push(`100% coverage is the goal — every untested path is a path where bugs hide and vibe coding becomes yolo coding. Evaluate what was ACTUALLY coded (from the diff), not what was planned.`);
} else if (mode === 'plan') {
sections.push(`100% coverage is the goal. Identify the tests each planned codepath needs. Add required proof for an exact approved behavior without asking again; take new policies or optional verification depth through the decision gate before treating their tests as accepted work. Review the requirements here; do not build the proposed tests.`);
sections.push(`${COVERAGE_GOAL} Evaluate what was ACTUALLY coded (from the diff), not what was planned.`);
} else {
sections.push(`100% coverage is the goal. Evaluate every codepath changed in the diff and identify test gaps. Gaps become INFORMATIONAL findings that follow the Fix-First flow.`);
sections.push(`${COVERAGE_GOAL} Identify the tests each planned codepath needs. Add required proof for an exact approved behavior without asking again; take new policies or optional verification depth through the decision gate before treating their tests as accepted work. Review the requirements here; do not build the proposed tests.`);
}
// ── Test framework detection (shared) ──
@@ -257,7 +279,7 @@ ls jest.config.* vitest.config.* playwright.config.* cypress.config.* .rspec pyt
git ls-files | grep -cE '(^|/)(tests?|spec|__tests__)/|(^|/)tests?\\.py$|(^|/)test_[^/]+\\.py$|_test\\.(go|py|rb|ts|js|exs)$|\\.(test|spec)\\.[jt]sx?$|_spec\\.rb$|Test\\.(java|kt)$' | sed 's/^/TESTFILES:/'
\`\`\`
3. **If no framework detected:**${mode === 'ship' ? ' use the bootstrap decision already made in Step 4; report diagram-only coverage if setup was declined. Do not restart bootstrap from this audit.' : mode === 'plan' ? ' State that the framework is unknown; continue the diagram and planned assertions. If proposing a new framework, settle that choice through Decision procedure in Test step 5. Reuse an exact prior approval; with no selection proposed, ask no framework question. Do not install a framework or write the proposed tests during this review.' : ' still produce the coverage diagram, but skip test generation.'}`);
3. **If no framework detected:**${mode === 'ship' ? ' use the bootstrap decision already made in Step 4; report diagram-only coverage if setup was declined. Do not restart bootstrap from this audit.' : ' State that the framework is unknown; continue the diagram and planned assertions. If proposing a new framework, settle that choice through Decision procedure in Test step 5. Reuse an exact prior approval; with no selection proposed, ask no framework question. Do not install a framework or write the proposed tests during this review.'}`);
// ── Before/after count (ship only) ──
if (mode === 'ship') {
@@ -277,7 +299,7 @@ Store this number for the PR body.`);
? `**Step 1. Trace every codepath in the plan:**
Read the plan document. For each new feature, service, endpoint, or component described, trace how data will flow through the code — don't just list planned functions, actually follow the planned execution:`
: `**${mode === 'ship' ? '1' : 'Step 1'}. Trace every codepath changed** using \`git diff origin/<base>${mode === 'ship' ? '' : '...HEAD'}\`:
: `**1. Trace every codepath changed** using \`git diff origin/<base>\`:
Read every changed file. For each one, trace how data flows through the code — don't just list functions, actually follow the execution:`;
@@ -357,7 +379,9 @@ Go through your diagram branch by branch — both code paths AND user flows. For
Quality scoring rubric:
- ★★★ Tests behavior with edge cases AND error paths
- ★★ Tests correct behavior, happy path only
- ★ Smoke test / existence check / trivial assertion (e.g., "it renders", "it doesn't throw")`);
- ★ Smoke test / existence check / trivial assertion (e.g., "it renders", "it doesn't throw"); weak, never counts as coverage
${generateTestValueBar(ctx, [mode])}`);
// ── E2E test decision matrix (shared) ──
sections.push(`
@@ -396,7 +420,9 @@ A regression is when:
- The existing test suite (if any) doesn't cover the changed path
- The change introduces a new failure mode for existing callers
When uncertain whether a change is a regression, err on the side of writing the test.${mode === 'review' ? '\n\nFormat: commit as `test: regression test for {what broke}`' : ''}`);
When uncertain whether a change is a regression, err on the side of writing the test.
${generateBaseControl(ctx)}`);
// ── ASCII coverage diagram (shared) ──
sections.push(`
@@ -432,7 +458,7 @@ Avoid bare \`[ ]\` or \`[x]\` in diagrams unless the block includes
\`Legend: [x] tested | [ ] no test\`. Prefer \`[GAP]\`, \`[★★ TESTED]\`,
\`[→E2E]\`, \`[→EVAL]\`; keep user-flow markers off code-path rows.
**Fast path:** All paths covered → "${mode === 'ship' ? 'Step 7' : mode === 'review' ? 'Step 4.75' : 'Test review'}: All new code paths have test coverage ✓" ${mode === 'plan' ? 'Still check LLM/eval scope and produce the Test Plan Artifact below.' : 'Continue.'}`);
**Fast path:** All paths covered → "${mode === 'ship' ? 'Step 7' : 'Test review'}: All new code paths have test coverage ✓" ${mode === 'plan' ? 'Still check LLM/eval scope and produce the Test Plan Artifact below.' : 'Continue.'}`);
// ── Mode-specific action section ──
if (mode === 'plan') {
@@ -447,8 +473,11 @@ Collect the requirements for each GAP and the LLM/eval scope above. Carry forwar
- What test file to create (match existing naming conventions)
- What the test should assert (specific inputs → expected outputs/behavior)
- Whether it's a unit test, E2E test, or eval (use the decision matrix)
- Its value card (test value bar above)
- For regression risks: flag as **CRITICAL** and name the behavior to protect
A proposal that fails the value bar becomes "extend <existing test>" or is dropped with a one-line reason. Also list **Tests made obsolete by this plan** (proposal only; retiring one still needs a complete retirement card at implementation time, see /test-audit).
Run the decision gate for this section's new or reopened choices. **STOP for each pending decision.** Wait for its answer before applying that remedy, moving to the next section or calling ExitPlanMode.
When these test and eval choices are resolved, write the Test Plan Artifact below. Its approved requirements should be specific enough to implement alongside the feature code.`);
@@ -486,17 +515,26 @@ Repo: {owner/repo}
## Critical Paths
- {end-to-end flow that must work}
Value: protects={...}; fails_when={...}; why_new={...}; seam=none
## Tests to Retire
- {existing test made obsolete by this plan and why, or none}
## Pending Decisions
- {unapproved test requirement and its ledger row, or none}
\`\`\`
Give each Edge Case and Critical Path entry its value card line. \`/test-audit\` reads \`## Tests to Retire\` from the newest artifact for the branch as seed candidates.
This file is consumed by \`/qa\` and \`/qa-only\` as primary test input. Include only the information that helps a QA tester know **what to test and where** — not implementation details.`);
} else if (mode === 'ship') {
sections.push(`
**5. Generate tests for uncovered paths:**
If test framework detected (or bootstrapped in Step 4):
- Apply the test value bar before writing each test. Extend an existing test (a new table row, fixture case or assertion) before creating a file. Record every proposal you decline in \`tests_rejected\` with a \`reason_code\` from \`${REASON_CODES.join(', ')}\`.
- Never add a production seam for a test; a seam that is not \`none\` names its non-test callers: \`seam=<name> (non-test callers: N, via <search command>)\`.
- Write the value card as a header comment in each generated or extended test.
- Prioritize error handlers and edge cases first (happy paths are more likely already tested)
- Read 2-3 existing test files to match conventions exactly
- Generate unit tests. Mock all external dependencies (DB, API, Redis).
@@ -506,7 +544,9 @@ If test framework detected (or bootstrapped in Step 4):
- Run each test. Passes → keep the change and report its path; the parent commits in Step 15.
- Fails → diagnose whether the test/fixture is invalid or a declared product contract is broken. Correct a demonstrated test defect once; preserve a valid red regression and route the reproduced product failure through the parent's fix/approval flow. Never delete or weaken it to manufacture green; retain unresolved coverage in the diagram.
Caps: 30 code paths max, 20 tests generated max (code + user flow combined), 2-min per-test exploration cap.
Caps: 30 code paths max; 5 tests per generation pass (code + user flow combined; the parent's \`Generation cap:\` overrides 5); an extension uses one slot and a rejection uses none; 2-min per-test exploration cap. List each remaining gap below the diagram (inside \`diagram\`) as a proposed test with its value card.
Do not rate stars for tests you wrote in this pass: count them as unrated (weak, reason \`unrated\`). The parent's read-only rating dispatch rates them. Counts are disjoint, precedence extended > added > rejected: one gap lands in at most one of \`tests_extended\`, \`tests_added\`, \`tests_rejected\`.
If no test framework AND user declined bootstrap → diagram only, no generation. Note: "Test generation skipped — no test framework configured."
@@ -520,41 +560,50 @@ git ls-files 2>/dev/null | grep -E '(\\.test\\.|\\.spec\\.|_test\\.|_spec\\.)' |
\`\`\`
For PR body: \`Tests: {before} → {after} (+{delta} new)\`
Coverage line: \`Test Coverage Audit: N new code paths. M covered (X%). K tests generated, awaiting parent commit.\``);
Coverage line: \`Test Coverage Audit: N new code paths. M covered (Y% any test, X% value-weighted). K tests generated, awaiting parent commit.\``);
gate = `
**7. Coverage gate:**
The parent owns this gate, including after inline fallback. Generated tests stay uncommitted until Step 15. Use Step 7's remaining generation allowance; supply it and the remaining gaps to the same audit prompt. At the cap, omit A and recommend stopping; the listed risk choices remain available.
The parent owns this gate, including after inline fallback. Generated tests stay uncommitted until Step 15. The gate only asks; it never hard-fails. Use Step 7's remaining generation allowance; supply it and the remaining gaps to the same audit prompt. At the cap, omit A's generation pass and recommend stopping; A then only lists proposals and the listed risk choices remain available.
Before proceeding, check CLAUDE.md for a \`## Test Coverage\` section with \`Minimum:\` and \`Target:\` fields. If found, use those percentages. Otherwise use defaults: Minimum = 60%, Target = 80%.
Read CLAUDE.md's \`## Test Coverage\` section for \`Minimum:\` and \`Target:\`; otherwise use defaults: Minimum = 60%, Target = 80%. Also read the optional \`Generation cap:\` (tests per pass, default 5), \`Base control:\` (\`auto\` default, or \`off\`), \`Base control budget:\` (seconds per run, default 90) and \`Star rating:\` (\`auto\` default, or \`off\`). Missing keys use the defaults.
Using the coverage percentage from the diagram in substep 4 (the \`COVERAGE: X/Y (Z%)\` line):
**Gate number X.** Take the first matching row; never substitute 0:
- **>= target:** Pass. "Coverage gate: PASS ({X}%)." Continue.
| Step 7 result | Gate number | Print |
|---|---|---|
| Rating dispatch failed or timed out | skip the gate | ${degradedMessage(ctx, 'ratingUnavailable')} |
| Zero paths, test-only diff, or \`coverage_pct\` null or unparseable | skip the gate | "Coverage gate: could not determine percentage — skipping." |
| \`Star rating: off\` | \`coverage_pct\` | "Star rating off: gate uses coverage_pct; weak paths still listed." |
| \`coverage_pct_value\` missing, not a number, or outside 0..100 | \`coverage_pct\` | ${degradedMessage(ctx, 'valueCoverageUnavailable')} |
| \`coverage_pct_value\` > \`coverage_pct\` | \`coverage_pct\` (clamped) | ${degradedMessage(ctx, 'inconsistentCoverage')} |
| Otherwise | \`coverage_pct_value\` | — |
Y is \`coverage_pct\`; W is \`weak_gaps.length\`; N is \`gaps\`. Remaining slots = 2 × generation cap − tests added or extended so far, and 0 once both passes are used. Option A reads "A) Strengthen the existing ★ test for each weak path and generate tests for true gaps ({slots} of {2 × cap} generation slots remaining)"; at 0 slots it reads "A) List the remaining gaps as proposed tests in the PR body" and dispatches nothing.
- **>= target:** Pass. "Coverage gate: PASS ({X}% value-weighted)." Continue; list weak paths in the PR body as proposed strengthening.
- **>= minimum, < target:** Use AskUserQuestion:
- "AI-assessed coverage is {X}%. {N} code paths are untested. Target is {target}%."
- RECOMMENDATION: Choose A because untested code paths are where production bugs hide.
- "Value-weighted coverage is {X}% ({Y}% including {W} weakly covered paths). {W} paths have only weak tests and {N} have none. Target is {target}%."
- RECOMMENDATION: Choose A because weakly covered and untested paths are where regressions slip through.
- Options:
A) Generate more tests for remaining gaps (recommended)
A) (as above, recommended)
B) Ship anyway — I accept the coverage risk
C) These paths don't need tests — mark as intentionally uncovered
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B/C or stop; never another generation pass.
C) These paths don't need tests — mark as intentionally uncovered. ${SWEEP_POINTER}
- If A and allowance remains: dispatch one generation pass with the weak paths and gaps, then re-evaluate here. At the cap, offer only B/C or stop, plus A as the proposals list; never another generation pass.
- If B: Continue. Include in PR body: "Coverage gate: {X}% — user accepted risk."
- If C: Continue. Include in PR body: "Coverage gate: {X}% — {N} paths intentionally uncovered."
- **< minimum:** Use AskUserQuestion:
- "AI-assessed coverage is critically low ({X}%). {N} of {M} code paths have no tests. Minimum threshold is {minimum}%."
- RECOMMENDATION: Choose A because less than {minimum}% means more code is untested than tested.
- "Value-weighted coverage is critically low ({X}%; {Y}% including {W} weakly covered paths). {N} of {M} code paths have no tests. Minimum threshold is {minimum}%."
- RECOMMENDATION: Choose A because less than {minimum}% means more behavior is unprotected than protected.
- Options:
A) Generate tests for remaining gaps (recommended)
A) (as above, recommended)
B) Override — ship with low coverage (I understand the risk)
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B or stop; never another generation pass.
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B or stop, plus A as the proposals list; never another generation pass.
- If B: Continue. Include in PR body: "Coverage gate: OVERRIDDEN at {X}%."
**Coverage percentage undetermined:** If the coverage diagram doesn't produce a clear numeric percentage (ambiguous output, parse error), **skip the gate** with: "Coverage gate: could not determine percentage — skipping." Do not default to 0% or block.
**Test-only diffs:** Skip the gate (same as the existing fast-path).
**Spawned or non-interactive session** (the preamble echoed \`SESSION_KIND: spawned\` or \`headless\`): ask nothing. Take A restricted to true \`gaps\` within the remaining slots; never edit tests for weak paths there. List weak paths in the PR body as proposed strengthening.
**100% coverage:** "Coverage gate: PASS (100%)." Continue.`;
@@ -590,51 +639,19 @@ Repo: {owner/repo}
## Critical Paths
- {end-to-end flow that must work}
\`\`\``);
} else {
// review mode
sections.push(`
**Step 5. Generate tests for gaps (Fix-First):**
If test framework is detected and gaps were identified:
- Classify each gap as AUTO-FIX or ASK per the Fix-First Heuristic:
- **AUTO-FIX:** Simple unit tests for pure functions, edge cases of existing tested functions
- **ASK:** E2E tests, tests requiring new test infrastructure, tests for ambiguous behavior
- For AUTO-FIX gaps: generate the test, run it, commit as \`test: coverage for {feature}\`
- For ASK gaps: include in the Fix-First batch question with the other review findings
- For paths marked [→E2E]: always ASK (E2E tests are higher-effort and need user confirmation)
- For paths marked [→EVAL]: always ASK (eval tests need user confirmation on quality criteria)
If no test framework detected → include gaps as INFORMATIONAL findings only, no generation.
**Diff is test-only changes:** Skip Step 4.75 entirely: "No new application code paths to audit."
### Coverage Warning
After producing the coverage diagram, check the coverage percentage. Read CLAUDE.md for a \`## Test Coverage\` section with a \`Minimum:\` field. If not found, use default: 60%.
If coverage is below the minimum threshold, output a prominent warning **before** the regular review findings:
\`\`\`
⚠️ COVERAGE WARNING: AI-assessed coverage is {X}%. {N} code paths untested.
Consider writing tests before running /ship.
\`\`\`
This is INFORMATIONAL — does not block /review. But it makes low coverage visible early so the developer can address it before reaching the /ship coverage gate.
If coverage percentage cannot be determined, skip the warning silently.`);
}
return part === 'gate' ? gate : sections.join('\n');
}
export function generateTestCoverageAuditPlan(_ctx: TemplateContext): string {
return generateTestCoverageAuditInner('plan');
export function generateTestCoverageAuditPlan(ctx: TemplateContext): string {
return generateTestCoverageAuditInner(ctx, 'plan');
}
export function generateTestCoverageAuditShip(_ctx: TemplateContext): string {
return generateTestCoverageAuditInner('ship');
export function generateTestCoverageAuditShip(ctx: TemplateContext): string {
return generateTestCoverageAuditInner(ctx, 'ship');
}
export function generateTestCoverageGateShip(_ctx: TemplateContext): string {
return generateTestCoverageAuditInner('ship', 'gate');
export function generateTestCoverageGateShip(ctx: TemplateContext): string {
return generateTestCoverageAuditInner(ctx, 'ship', 'gate');
}