mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
Merge origin/main (v1.91.9.0, #2998) into capy/audit-fix-wave; release as v1.91.10.0
Main shipped v1.91.9.0, so this wave becomes v1.91.10.0. Conflicts kept this branch's planner-derived assertions. Main's new test-value eval gets rule kinds for its three gate cases and its measured 332 s duration from #2998's CI; census counts and the PR fallback floor follow the new file. The packing test now weighs each tier's recorded durations the way the planner does.
This commit is contained in:
commit
58b5e3f977
58 files changed
+2453
-220
No files matched your search
@@ -39,6 +39,7 @@ import { generateAsideSetup, generateAsideCookbook, generateAsideResearch, gener
|
||||
import { generateCommandReference, generateSnapshotFlags, generateBrowseSetup, generateBrowseFallback } from './browse';
|
||||
import { generateDesignDocDiscovery } from './design-doc-discovery';
|
||||
import { generateSharedLibsRubric } from './shared-libs';
|
||||
import { generateTestValueBar, generateTestValueMessage } from './test-value';
|
||||
import { generateQAScope, generateQAExploratory, generateQAFunctional, generateQAResource, generateQAReview, generateQAReviewPreflight, generateQAMethodReads } from './qa';
|
||||
|
||||
export const RESOLVERS: Record<string, ResolverFn> = {
|
||||
@@ -100,6 +101,8 @@ export const RESOLVERS: Record<string, ResolverFn> = {
|
||||
TEST_COVERAGE_AUDIT_PLAN: generateTestCoverageAuditPlan,
|
||||
TEST_COVERAGE_AUDIT_SHIP: generateTestCoverageAuditShip,
|
||||
TEST_COVERAGE_GATE_SHIP: generateTestCoverageGateShip,
|
||||
TEST_VALUE_BAR: generateTestValueBar,
|
||||
TEST_VALUE_MESSAGE: generateTestValueMessage,
|
||||
TEST_FAILURE_TRIAGE: generateTestFailureTriage,
|
||||
SPEC_REVIEW_LOOP: generateSpecReviewLoop,
|
||||
DESIGN_SKETCH: generateDesignSketch,
|
||||
|
||||
@@ -0,0 +1,171 @@
|
||||
import type { TemplateContext } from './types';
|
||||
|
||||
// Test value bar shared by plan-eng-review and ship (through the coverage audit),
|
||||
// qa/qa-only ({{TEST_VALUE_BAR:qa}}) and test-audit ({{TEST_VALUE_BAR:audit}}).
|
||||
// The static review/specialists/testing.md repeats these constants and is kept
|
||||
// in sync by test/test-value-bar.test.ts.
|
||||
// Adapted from openclaw/openclaw@a214e76, .agents/skills/test-audit/SKILL.md.
|
||||
|
||||
export const TEST_VALUE_BAR_MODES = ['plan', 'ship', 'qa', 'audit'] as const;
|
||||
export type TestValueBarMode = (typeof TEST_VALUE_BAR_MODES)[number];
|
||||
|
||||
export const QUESTIONS = [
|
||||
'What observable behavior, invariant or independent contract does it protect?',
|
||||
'What credible regression makes it fail?',
|
||||
'Why does existing coverage not already catch that? Prefer adding a row to an existing table-driven test or shared fixture over a near-duplicate.',
|
||||
'Does it need a production seam (export, flag, wrapper, injection hook) that no production caller needs? If yes, test at the real boundary instead.',
|
||||
] as const;
|
||||
|
||||
export const VALUE_CARD_FIELDS = ['protects', 'fails_when', 'why_new', 'seam'] as const;
|
||||
export const CARD_FIELD_MAX_BYTES = 160;
|
||||
|
||||
export const CATALOG = [
|
||||
'assertion-free coverage probes',
|
||||
'self-comparisons and identity copies',
|
||||
'copied fixtures, inventories or export lists',
|
||||
'exact source, import or string greps that are not a declared contract',
|
||||
'private predicate or call-shape tests duplicated at a real boundary',
|
||||
'duplicate invocations of the same contract',
|
||||
"per-caller replays of a shared helper's tests",
|
||||
'tests whose only purpose is keeping a test-only export, global or wrapper alive',
|
||||
'production code whose only callers are tests',
|
||||
] as const;
|
||||
|
||||
export const RETIREMENT_FIELDS = ['test', 'detects', 'non_test_callers', 'search_command', 'stronger_proof', 'history', 'unlocks', 'validation'] as const;
|
||||
export const REVIEW_EVIDENCE_FIELDS = ['detects', 'non_test_callers', 'search_command', 'stronger_proof'] as const;
|
||||
export const REASON_CODES = ['duplicate_protects', 'needs_seam', 'incomplete_card', 'no_credible_regression', 'covered_elsewhere', 'implementation_coupled'] as const;
|
||||
export const WEAK_REASONS = ['star_one', 'gate_failed', 'unrated'] as const;
|
||||
export const PRAGMA = 'gstack:test-value keep';
|
||||
export const SWEEP_POINTER = 'Repo-wide sweep: run /test-audit.';
|
||||
|
||||
export const RETENTION_ONE_LINER = 'Retention bar: keep a test that independently enforces a public API, protocol, config, migration, storage, security, platform, default, prompt-byte, generated-output (golden), package, release or architecture contract; static or slow is no reason to delete.';
|
||||
|
||||
export const CALLER_SEARCH_COMMAND = "git grep -n -F -w -e '<symbol>' -- . ':!test/' ':!tests/' ':!spec/' ':!**/__tests__/**' ':!**/*.test.*' ':!**/*.spec.*' ':!**/*_test.*' ':!**/test_*.py'";
|
||||
export const CALLER_SYMBOL_PATTERN = '^[A-Za-z_][A-Za-z0-9_]*$';
|
||||
|
||||
export const DOCS_PAGE = 'docs/test-value-bar.md';
|
||||
|
||||
export const MESSAGES = {
|
||||
ratingUnavailable: {
|
||||
message: 'rating unavailable: the read-only rating dispatch failed or timed out, so the coverage gate is skipped for this run. Re-run Step 7 to re-rate the tests.',
|
||||
anchor: 'rating-unavailable',
|
||||
},
|
||||
valueCoverageUnavailable: {
|
||||
message: 'value-weighted coverage unavailable (outdated installed skill); run /gstack-upgrade. The gate used coverage_pct (any test) this run.',
|
||||
anchor: 'value-weighted-coverage-unavailable',
|
||||
},
|
||||
inconsistentCoverage: {
|
||||
message: 'inconsistent coverage inputs: coverage_pct_value was above coverage_pct, so it was clamped to coverage_pct. Re-run Step 7 if the numbers look wrong.',
|
||||
anchor: 'inconsistent-coverage-inputs',
|
||||
},
|
||||
malformedKey: {
|
||||
message: 'malformed <key> ignored: the audit returned the wrong type, so it counts as empty. The likely cause is an outdated installed skill; run /gstack-upgrade.',
|
||||
anchor: 'malformed-key-ignored',
|
||||
},
|
||||
allRejected: {
|
||||
message: 'all <N> generated tests rejected by machine checks; see tests_rejected. The gate proceeds with the unchanged value-weighted coverage.',
|
||||
anchor: 'all-generated-tests-rejected',
|
||||
},
|
||||
baseControlUnavailable: {
|
||||
message: 'base control unavailable: <reason>. The fails-at-HEAD result still stands. To check by hand: `git worktree add --detach <tmp> <base>`; copy the test and its new fixtures to the same paths; run the detected test command in <tmp>; `git worktree remove --force <tmp>`. Then report `passes at base: manual`.',
|
||||
anchor: 'base-control-unavailable',
|
||||
},
|
||||
callerCheckUnavailable: {
|
||||
message: 'caller check unavailable: <command or "unsupported symbol">. The finding stays INFORMATIONAL and nothing is proposed for deletion; run the search by hand to complete the evidence.',
|
||||
anchor: 'caller-check-unavailable',
|
||||
},
|
||||
unknownMode: {
|
||||
message: 'Unknown TEST_VALUE_BAR mode <x>; expected plan|ship|qa|audit. Fix the placeholder or add the mode in scripts/resolvers/test-value.ts.',
|
||||
anchor: 'unknown-test-value-bar-mode',
|
||||
},
|
||||
} as const;
|
||||
|
||||
export type MessageKey = keyof typeof MESSAGES;
|
||||
|
||||
export function degradedMessage(ctx: TemplateContext, key: MessageKey): string {
|
||||
const { message, anchor } = MESSAGES[key];
|
||||
return `${message} (see ${ctx.paths.skillRoot}/${DOCS_PAGE}#${anchor})`;
|
||||
}
|
||||
|
||||
export function generateTestValueMessage(ctx: TemplateContext, args?: string[]): string {
|
||||
const key = args?.[0] ?? '';
|
||||
if (!(key in MESSAGES)) throw new Error(`Unknown TEST_VALUE_MESSAGE key ${key}; expected ${Object.keys(MESSAGES).join('|')} (scripts/resolvers/test-value.ts)`);
|
||||
return degradedMessage(ctx, key as MessageKey);
|
||||
}
|
||||
|
||||
// Measured renders plus 15%: plan 1897, ship 2742, qa 1036, audit 3712 bytes.
|
||||
export const TEST_VALUE_BAR_MAX_BYTES: Record<TestValueBarMode, number> = { plan: 2182, ship: 3154, qa: 1192, audit: 4269 };
|
||||
|
||||
export function clampCardField(value: string): string {
|
||||
const bytes = Buffer.from(value, 'utf8');
|
||||
if (bytes.length <= CARD_FIELD_MAX_BYTES) return value;
|
||||
let end = CARD_FIELD_MAX_BYTES - 3;
|
||||
while (end > 0 && (bytes[end]! & 0xc0) === 0x80) end--;
|
||||
return `${bytes.subarray(0, end).toString('utf8')}...`;
|
||||
}
|
||||
|
||||
export function renderValueCard(card: Record<(typeof VALUE_CARD_FIELDS)[number], string>): string {
|
||||
return `Value: ${VALUE_CARD_FIELDS.map(field => `${field}=${clampCardField(card[field])}`).join('; ')}`;
|
||||
}
|
||||
|
||||
const EXAMPLE_CARD = renderValueCard({
|
||||
protects: 'refundPayment rejects an empty reason',
|
||||
fails_when: 'the reason guard is removed or inverted',
|
||||
why_new: 'billing.test.ts covers processPayment only',
|
||||
seam: 'none',
|
||||
});
|
||||
|
||||
const EXAMPLE_REJECTED = 'Rejected (covered_elsewhere): "checkout renders"; checkout.e2e.ts:15 covers it, so extend that test.';
|
||||
|
||||
function questionList(mode: TestValueBarMode): string {
|
||||
const numbered = QUESTIONS.map((question, index) => `${index + 1}. ${question}`);
|
||||
return mode === 'qa' ? numbered.slice(2).join('\n') : numbered.join('\n');
|
||||
}
|
||||
|
||||
function cardRules(mode: TestValueBarMode): string {
|
||||
const where = {
|
||||
plan: 'One card per Critical Path and Edge Case in the Test Plan Artifact.',
|
||||
ship: 'Write it as a header comment in each generated test, next to the attribution (wrap, do not truncate); with no known comment syntax, put it in the PR body\'s Test value details.',
|
||||
qa: 'Put it in the 8e.5 record (/qa) or under each proposed test (/qa-only).',
|
||||
audit: 'Read cards from test header comments when present.',
|
||||
}[mode];
|
||||
return `Value card: \`Value: protects=<...>; fails_when=<...>; why_new=<...>; seam=none\` (seam: \`none\` or its name); each field at most ${CARD_FIELD_MAX_BYTES} UTF-8 bytes here (clamp to 157 plus \`...\`; JSON keeps full values). ${where} A missing upstream card never blocks: derive it; ignore unknown fields.
|
||||
|
||||
Example: ${EXAMPLE_CARD}
|
||||
${EXAMPLE_REJECTED}`;
|
||||
}
|
||||
|
||||
const STAR_RULE = 'Weak tests (★ smoke/existence/trivial, gate-failing or unrated) never count as coverage. X = paths with a ★★/★★★ test / total paths (value-weighted; the gate uses X); Y = paths with any test / total paths.';
|
||||
|
||||
const WEAK_PATH_RULE = `Total paths = the diff's codepath trace, max 30; zero skips the gate. A path with only weak tests is uncovered in X, covered in Y, and goes to \`weak_gaps\` (reason \`${WEAK_REASONS.join('|')}\`), not \`gaps\`. Rate stars only for tests reachable from changed paths.`;
|
||||
|
||||
const RED_PROOF = `Regression proof: a regression test must fail at HEAD before any repair, in its own assertion (a pass at HEAD drops the regression label; an import, fixture or env failure is a test defect: correct once or drop). It must pass at base as the control (an assertion failure there marks it invalid; any other failure is "base control unavailable: collection error") and pass after the repair. Record: \`Regression proof — fails at HEAD: yes · passes at base: yes | unavailable (<reason>) | manual · passes after fix: yes | pending\`.`;
|
||||
|
||||
function auditSections(): string {
|
||||
return `Low-value catalog (a match fails the gate unless the retention bar names the contract it guards):
|
||||
${CATALOG.map(entry => `- ${entry}`).join('\n')}
|
||||
|
||||
Retention bar: keep a test that independently enforces a public API, protocol, config, migration, storage, security, platform, default, prompt-byte, generated-output (SKILL.md golden), package, release or architecture contract; call order when order is observable; source inspection when it is the cheapest independent guard. Never retire anything reachable from the package entrypoint (\`package.json\` exports/main, index re-exports). Static or slow is not a reason to delete. Skip a test carrying \`${PRAGMA} reason="<why>"\` and list it as suppressed.
|
||||
|
||||
Retirement card, complete before any edit: ${RETIREMENT_FIELDS.map(field => `\`${field}\``).join(', ')}. Caller check for a symbol matching \`${CALLER_SYMBOL_PATTERN}\` (otherwise "caller check unavailable: unsupported symbol"): \`${CALLER_SEARCH_COMMAND}\`; record the command, exclusions and hit count. The evidence is grep-only (no re-exports, dynamic dispatch or generated code), so production code is retired only when the repo's typecheck/build or dead-code tool passes with it removed in a scratch worktree.`;
|
||||
}
|
||||
|
||||
export function generateTestValueBar(_ctx: TemplateContext, args?: string[]): string {
|
||||
const mode = args?.[0] as TestValueBarMode;
|
||||
if (!TEST_VALUE_BAR_MODES.includes(mode)) throw new Error(MESSAGES.unknownMode.message.replace('<x>', String(args?.[0])));
|
||||
const parts = [
|
||||
`**Test value bar.** ${mode === 'qa' ? 'Before writing the test (the reproduced bug answers what it protects and what makes it fail):' : 'Propose or write a test only with all four answers; otherwise extend an existing test or drop it:'}`,
|
||||
questionList(mode),
|
||||
cardRules(mode),
|
||||
];
|
||||
if (mode !== 'qa') parts.splice(2, 0, 'A test that breaks under a behavior-preserving refactor asserts implementation: rewrite it at the owning boundary, unless exact output is the declared contract (goldens, prompt bytes, wire formats).');
|
||||
if (mode === 'plan') parts.push(`${STAR_RULE} /ship computes them; here every proposed test needs a card.`, RETENTION_ONE_LINER);
|
||||
if (mode === 'ship') parts.push(`${STAR_RULE} ${WEAK_PATH_RULE}`, RETENTION_ONE_LINER);
|
||||
if (mode === 'ship' || mode === 'audit') parts.push(RED_PROOF);
|
||||
if (mode === 'audit') parts.push(auditSections());
|
||||
const rendered = parts.join('\n\n');
|
||||
const bytes = Buffer.byteLength(rendered, 'utf8');
|
||||
const budget = TEST_VALUE_BAR_MAX_BYTES[mode];
|
||||
if (bytes > budget) throw new Error(`TEST_VALUE_BAR mode '${mode}' renders ${bytes} bytes, budget ${budget} (over by ${bytes - budget}). Trim the mode's section in scripts/resolvers/test-value.ts or raise the ceiling with a reason.`);
|
||||
return rendered;
|
||||
}
|
||||
@@ -1,5 +1,6 @@
|
||||
import type { TemplateContext } from './types';
|
||||
import { asideExecPrelude } from './aside';
|
||||
import { generateTestValueBar, degradedMessage, REASON_CODES, SWEEP_POINTER, type TestValueBarMode } from './test-value';
|
||||
|
||||
export function generateTestBootstrap(ctx: TemplateContext): string {
|
||||
return `## Test Framework Bootstrap
|
||||
@@ -196,38 +197,59 @@ Only commit if there are changes. Stage all bootstrap files (config, test direct
|
||||
// ─── Test Coverage Audit ────────────────────────────────────
|
||||
//
|
||||
// Shared methodology for codepath tracing, ASCII diagrams, and test gap analysis.
|
||||
// Three modes, three placeholders, one inner function:
|
||||
// Two modes, one inner function; both embed the test value bar (test-value.ts):
|
||||
//
|
||||
// {{TEST_COVERAGE_AUDIT_PLAN}} → plan-eng-review: adds missing tests to the plan
|
||||
// {{TEST_COVERAGE_AUDIT_SHIP}} → ship: auto-generates tests, coverage summary
|
||||
// {{TEST_COVERAGE_AUDIT_REVIEW}} → review: generates tests via Fix-First (ASK)
|
||||
// {{TEST_COVERAGE_AUDIT_SHIP}} → ship: generates tests, coverage summary
|
||||
// {{TEST_COVERAGE_GATE_SHIP}} → ship: parent-owned coverage gate
|
||||
//
|
||||
// ┌────────────────────────────────────────────────┐
|
||||
// │ generateTestCoverageAuditInner(mode) │
|
||||
// │ │
|
||||
// │ SHARED: framework detect, codepath trace, │
|
||||
// │ ASCII diagram, quality rubric, E2E matrix, │
|
||||
// │ regression rule │
|
||||
// │ │
|
||||
// │ plan: edit plan file, write artifact │
|
||||
// │ ship: auto-generate tests, write artifact │
|
||||
// │ review: Fix-First ASK, INFORMATIONAL gaps │
|
||||
// └────────────────────────────────────────────────┘
|
||||
// /review uses its static testing specialist (review/specialists/testing.md).
|
||||
|
||||
type CoverageAuditMode = 'plan' | 'ship' | 'review';
|
||||
export type CoverageAuditMode = Extract<TestValueBarMode, 'plan' | 'ship'>;
|
||||
|
||||
function generateTestCoverageAuditInner(mode: CoverageAuditMode, part: 'audit' | 'gate' = 'audit'): string {
|
||||
function generateBaseControl(ctx: TemplateContext): string {
|
||||
return `**Red-first proof.** Apply the value bar's Regression proof to every regression test: the diff at HEAD is the pre-fix code, so run the new test at HEAD before any repair. Then, unless the parent says \`Base control: off\`, run this base control once per regression test in diff order, within a 3-minute total per /ship run (\`Base control budget:\` seconds per test, default 90). Past the total, record "base control unavailable: budget" and report "N of M regression tests got base control". The block is one shell invocation; it installs nothing and runs no build or postinstall.
|
||||
|
||||
\`\`\`bash
|
||||
# Set: BASE = the base branch this /ship run resolved; TEST = the test file; FIXTURES = new
|
||||
# test-only fixtures it imports (repo-relative, may be empty); RUN = the detected
|
||||
# runner for one file (e.g. "bun test $TEST"); BUDGET = seconds for this run (default 90).
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
CTL_TMP=$(mktemp -d "\${TMPDIR:-/tmp}/gstack-base-control.XXXXXX")
|
||||
cleanup() { git -C "$ROOT" worktree remove --force "$CTL_TMP/wt" >/dev/null 2>&1; rm -rf "$CTL_TMP"; [ -e "$CTL_TMP" ] && echo "BASE_CONTROL_LEFTOVER: $CTL_TMP (run: git worktree prune)"; }
|
||||
trap cleanup EXIT INT TERM
|
||||
(
|
||||
[ -f "$ROOT/package.json" ] || { echo "BASE_CONTROL: unavailable (ecosystem)"; exit 0; }
|
||||
git -C "$ROOT" remote get-url origin >/dev/null 2>&1 || { echo "BASE_CONTROL: unavailable (no base remote)"; exit 0; }
|
||||
timeout 30 git -C "$ROOT" fetch --quiet origin "$BASE" || { echo "BASE_CONTROL: unavailable (base not fetched)"; exit 0; }
|
||||
git -C "$ROOT" worktree add --quiet --detach "$CTL_TMP/wt" "origin/$BASE" >/dev/null 2>&1 || { echo "BASE_CONTROL: unavailable (worktree add failed)"; exit 0; }
|
||||
for f in $TEST $FIXTURES; do mkdir -p "$CTL_TMP/wt/$(dirname "$f")" && cp "$ROOT/$f" "$CTL_TMP/wt/$f"; done
|
||||
[ -d "$ROOT/node_modules" ] && ln -s "$ROOT/node_modules" "$CTL_TMP/wt/node_modules"
|
||||
cd "$CTL_TMP/wt" && timeout "\${BUDGET:-90}" sh -c "$RUN" > "$CTL_TMP/out" 2>&1; rc=$?
|
||||
tail -n 40 "$CTL_TMP/out"
|
||||
if [ "$rc" -eq 0 ]; then echo "BASE_CONTROL: passes at base"
|
||||
elif [ "$rc" -eq 124 ]; then echo "BASE_CONTROL: unavailable (budget)"
|
||||
else echo "BASE_CONTROL: fails at base (exit $rc)"; fi
|
||||
)
|
||||
\`\`\`
|
||||
|
||||
Classify "fails at base" from the output: a failure in the test's own assertion marks it invalid (correct once or drop it); an import, collection, missing generated artifact or dependency failure is "base control unavailable: collection error" and the test stays. Print each unavailable result as: ${degradedMessage(ctx, 'baseControlUnavailable')}
|
||||
|
||||
Return \`"regression_proof":{"red_at_head":N,"base_green":N,"base_unavailable":N}\` counts in the JSON and each test's record line in the diagram.`;
|
||||
}
|
||||
|
||||
const COVERAGE_GOAL = 'Coverage goal: every changed behavior is protected by a test that would catch a real regression. Test count is not a goal.';
|
||||
|
||||
export function generateTestCoverageAuditInner(ctx: TemplateContext, mode: CoverageAuditMode, part: 'audit' | 'gate' = 'audit'): string {
|
||||
const sections: string[] = [];
|
||||
const subheading = mode === 'plan' ? '####' : '###';
|
||||
let gate = '';
|
||||
|
||||
// ── Intro (mode-specific) ──
|
||||
if (mode === 'ship') {
|
||||
sections.push(`100% coverage is the goal — every untested path is a path where bugs hide and vibe coding becomes yolo coding. Evaluate what was ACTUALLY coded (from the diff), not what was planned.`);
|
||||
} else if (mode === 'plan') {
|
||||
sections.push(`100% coverage is the goal. Identify the tests each planned codepath needs. Add required proof for an exact approved behavior without asking again; take new policies or optional verification depth through the decision gate before treating their tests as accepted work. Review the requirements here; do not build the proposed tests.`);
|
||||
sections.push(`${COVERAGE_GOAL} Evaluate what was ACTUALLY coded (from the diff), not what was planned.`);
|
||||
} else {
|
||||
sections.push(`100% coverage is the goal. Evaluate every codepath changed in the diff and identify test gaps. Gaps become INFORMATIONAL findings that follow the Fix-First flow.`);
|
||||
sections.push(`${COVERAGE_GOAL} Identify the tests each planned codepath needs. Add required proof for an exact approved behavior without asking again; take new policies or optional verification depth through the decision gate before treating their tests as accepted work. Review the requirements here; do not build the proposed tests.`);
|
||||
}
|
||||
|
||||
// ── Test framework detection (shared) ──
|
||||
@@ -257,7 +279,7 @@ ls jest.config.* vitest.config.* playwright.config.* cypress.config.* .rspec pyt
|
||||
git ls-files | grep -cE '(^|/)(tests?|spec|__tests__)/|(^|/)tests?\\.py$|(^|/)test_[^/]+\\.py$|_test\\.(go|py|rb|ts|js|exs)$|\\.(test|spec)\\.[jt]sx?$|_spec\\.rb$|Test\\.(java|kt)$' | sed 's/^/TESTFILES:/'
|
||||
\`\`\`
|
||||
|
||||
3. **If no framework detected:**${mode === 'ship' ? ' use the bootstrap decision already made in Step 4; report diagram-only coverage if setup was declined. Do not restart bootstrap from this audit.' : mode === 'plan' ? ' State that the framework is unknown; continue the diagram and planned assertions. If proposing a new framework, settle that choice through Decision procedure in Test step 5. Reuse an exact prior approval; with no selection proposed, ask no framework question. Do not install a framework or write the proposed tests during this review.' : ' still produce the coverage diagram, but skip test generation.'}`);
|
||||
3. **If no framework detected:**${mode === 'ship' ? ' use the bootstrap decision already made in Step 4; report diagram-only coverage if setup was declined. Do not restart bootstrap from this audit.' : ' State that the framework is unknown; continue the diagram and planned assertions. If proposing a new framework, settle that choice through Decision procedure in Test step 5. Reuse an exact prior approval; with no selection proposed, ask no framework question. Do not install a framework or write the proposed tests during this review.'}`);
|
||||
|
||||
// ── Before/after count (ship only) ──
|
||||
if (mode === 'ship') {
|
||||
@@ -277,7 +299,7 @@ Store this number for the PR body.`);
|
||||
? `**Step 1. Trace every codepath in the plan:**
|
||||
|
||||
Read the plan document. For each new feature, service, endpoint, or component described, trace how data will flow through the code — don't just list planned functions, actually follow the planned execution:`
|
||||
: `**${mode === 'ship' ? '1' : 'Step 1'}. Trace every codepath changed** using \`git diff origin/<base>${mode === 'ship' ? '' : '...HEAD'}\`:
|
||||
: `**1. Trace every codepath changed** using \`git diff origin/<base>\`:
|
||||
|
||||
Read every changed file. For each one, trace how data flows through the code — don't just list functions, actually follow the execution:`;
|
||||
|
||||
@@ -357,7 +379,9 @@ Go through your diagram branch by branch — both code paths AND user flows. For
|
||||
Quality scoring rubric:
|
||||
- ★★★ Tests behavior with edge cases AND error paths
|
||||
- ★★ Tests correct behavior, happy path only
|
||||
- ★ Smoke test / existence check / trivial assertion (e.g., "it renders", "it doesn't throw")`);
|
||||
- ★ Smoke test / existence check / trivial assertion (e.g., "it renders", "it doesn't throw"); weak, never counts as coverage
|
||||
|
||||
${generateTestValueBar(ctx, [mode])}`);
|
||||
|
||||
// ── E2E test decision matrix (shared) ──
|
||||
sections.push(`
|
||||
@@ -396,7 +420,9 @@ A regression is when:
|
||||
- The existing test suite (if any) doesn't cover the changed path
|
||||
- The change introduces a new failure mode for existing callers
|
||||
|
||||
When uncertain whether a change is a regression, err on the side of writing the test.${mode === 'review' ? '\n\nFormat: commit as `test: regression test for {what broke}`' : ''}`);
|
||||
When uncertain whether a change is a regression, err on the side of writing the test.
|
||||
|
||||
${generateBaseControl(ctx)}`);
|
||||
|
||||
// ── ASCII coverage diagram (shared) ──
|
||||
sections.push(`
|
||||
@@ -432,7 +458,7 @@ Avoid bare \`[ ]\` or \`[x]\` in diagrams unless the block includes
|
||||
\`Legend: [x] tested | [ ] no test\`. Prefer \`[GAP]\`, \`[★★ TESTED]\`,
|
||||
\`[→E2E]\`, \`[→EVAL]\`; keep user-flow markers off code-path rows.
|
||||
|
||||
**Fast path:** All paths covered → "${mode === 'ship' ? 'Step 7' : mode === 'review' ? 'Step 4.75' : 'Test review'}: All new code paths have test coverage ✓" ${mode === 'plan' ? 'Still check LLM/eval scope and produce the Test Plan Artifact below.' : 'Continue.'}`);
|
||||
**Fast path:** All paths covered → "${mode === 'ship' ? 'Step 7' : 'Test review'}: All new code paths have test coverage ✓" ${mode === 'plan' ? 'Still check LLM/eval scope and produce the Test Plan Artifact below.' : 'Continue.'}`);
|
||||
|
||||
// ── Mode-specific action section ──
|
||||
if (mode === 'plan') {
|
||||
@@ -447,8 +473,11 @@ Collect the requirements for each GAP and the LLM/eval scope above. Carry forwar
|
||||
- What test file to create (match existing naming conventions)
|
||||
- What the test should assert (specific inputs → expected outputs/behavior)
|
||||
- Whether it's a unit test, E2E test, or eval (use the decision matrix)
|
||||
- Its value card (test value bar above)
|
||||
- For regression risks: flag as **CRITICAL** and name the behavior to protect
|
||||
|
||||
A proposal that fails the value bar becomes "extend <existing test>" or is dropped with a one-line reason. Also list **Tests made obsolete by this plan** (proposal only; retiring one still needs a complete retirement card at implementation time, see /test-audit).
|
||||
|
||||
Run the decision gate for this section's new or reopened choices. **STOP for each pending decision.** Wait for its answer before applying that remedy, moving to the next section or calling ExitPlanMode.
|
||||
|
||||
When these test and eval choices are resolved, write the Test Plan Artifact below. Its approved requirements should be specific enough to implement alongside the feature code.`);
|
||||
@@ -486,17 +515,26 @@ Repo: {owner/repo}
|
||||
|
||||
## Critical Paths
|
||||
- {end-to-end flow that must work}
|
||||
Value: protects={...}; fails_when={...}; why_new={...}; seam=none
|
||||
|
||||
## Tests to Retire
|
||||
- {existing test made obsolete by this plan and why, or none}
|
||||
|
||||
## Pending Decisions
|
||||
- {unapproved test requirement and its ledger row, or none}
|
||||
\`\`\`
|
||||
|
||||
Give each Edge Case and Critical Path entry its value card line. \`/test-audit\` reads \`## Tests to Retire\` from the newest artifact for the branch as seed candidates.
|
||||
|
||||
This file is consumed by \`/qa\` and \`/qa-only\` as primary test input. Include only the information that helps a QA tester know **what to test and where** — not implementation details.`);
|
||||
} else if (mode === 'ship') {
|
||||
sections.push(`
|
||||
**5. Generate tests for uncovered paths:**
|
||||
|
||||
If test framework detected (or bootstrapped in Step 4):
|
||||
- Apply the test value bar before writing each test. Extend an existing test (a new table row, fixture case or assertion) before creating a file. Record every proposal you decline in \`tests_rejected\` with a \`reason_code\` from \`${REASON_CODES.join(', ')}\`.
|
||||
- Never add a production seam for a test; a seam that is not \`none\` names its non-test callers: \`seam=<name> (non-test callers: N, via <search command>)\`.
|
||||
- Write the value card as a header comment in each generated or extended test.
|
||||
- Prioritize error handlers and edge cases first (happy paths are more likely already tested)
|
||||
- Read 2-3 existing test files to match conventions exactly
|
||||
- Generate unit tests. Mock all external dependencies (DB, API, Redis).
|
||||
@@ -506,7 +544,9 @@ If test framework detected (or bootstrapped in Step 4):
|
||||
- Run each test. Passes → keep the change and report its path; the parent commits in Step 15.
|
||||
- Fails → diagnose whether the test/fixture is invalid or a declared product contract is broken. Correct a demonstrated test defect once; preserve a valid red regression and route the reproduced product failure through the parent's fix/approval flow. Never delete or weaken it to manufacture green; retain unresolved coverage in the diagram.
|
||||
|
||||
Caps: 30 code paths max, 20 tests generated max (code + user flow combined), 2-min per-test exploration cap.
|
||||
Caps: 30 code paths max; 5 tests per generation pass (code + user flow combined; the parent's \`Generation cap:\` overrides 5); an extension uses one slot and a rejection uses none; 2-min per-test exploration cap. List each remaining gap below the diagram (inside \`diagram\`) as a proposed test with its value card.
|
||||
|
||||
Do not rate stars for tests you wrote in this pass: count them as unrated (weak, reason \`unrated\`). The parent's read-only rating dispatch rates them. Counts are disjoint, precedence extended > added > rejected: one gap lands in at most one of \`tests_extended\`, \`tests_added\`, \`tests_rejected\`.
|
||||
|
||||
If no test framework AND user declined bootstrap → diagram only, no generation. Note: "Test generation skipped — no test framework configured."
|
||||
|
||||
@@ -520,41 +560,50 @@ git ls-files 2>/dev/null | grep -E '(\\.test\\.|\\.spec\\.|_test\\.|_spec\\.)' |
|
||||
\`\`\`
|
||||
|
||||
For PR body: \`Tests: {before} → {after} (+{delta} new)\`
|
||||
Coverage line: \`Test Coverage Audit: N new code paths. M covered (X%). K tests generated, awaiting parent commit.\``);
|
||||
Coverage line: \`Test Coverage Audit: N new code paths. M covered (Y% any test, X% value-weighted). K tests generated, awaiting parent commit.\``);
|
||||
|
||||
gate = `
|
||||
**7. Coverage gate:**
|
||||
|
||||
The parent owns this gate, including after inline fallback. Generated tests stay uncommitted until Step 15. Use Step 7's remaining generation allowance; supply it and the remaining gaps to the same audit prompt. At the cap, omit A and recommend stopping; the listed risk choices remain available.
|
||||
The parent owns this gate, including after inline fallback. Generated tests stay uncommitted until Step 15. The gate only asks; it never hard-fails. Use Step 7's remaining generation allowance; supply it and the remaining gaps to the same audit prompt. At the cap, omit A's generation pass and recommend stopping; A then only lists proposals and the listed risk choices remain available.
|
||||
|
||||
Before proceeding, check CLAUDE.md for a \`## Test Coverage\` section with \`Minimum:\` and \`Target:\` fields. If found, use those percentages. Otherwise use defaults: Minimum = 60%, Target = 80%.
|
||||
Read CLAUDE.md's \`## Test Coverage\` section for \`Minimum:\` and \`Target:\`; otherwise use defaults: Minimum = 60%, Target = 80%. Also read the optional \`Generation cap:\` (tests per pass, default 5), \`Base control:\` (\`auto\` default, or \`off\`), \`Base control budget:\` (seconds per run, default 90) and \`Star rating:\` (\`auto\` default, or \`off\`). Missing keys use the defaults.
|
||||
|
||||
Using the coverage percentage from the diagram in substep 4 (the \`COVERAGE: X/Y (Z%)\` line):
|
||||
**Gate number X.** Take the first matching row; never substitute 0:
|
||||
|
||||
- **>= target:** Pass. "Coverage gate: PASS ({X}%)." Continue.
|
||||
| Step 7 result | Gate number | Print |
|
||||
|---|---|---|
|
||||
| Rating dispatch failed or timed out | skip the gate | ${degradedMessage(ctx, 'ratingUnavailable')} |
|
||||
| Zero paths, test-only diff, or \`coverage_pct\` null or unparseable | skip the gate | "Coverage gate: could not determine percentage — skipping." |
|
||||
| \`Star rating: off\` | \`coverage_pct\` | "Star rating off: gate uses coverage_pct; weak paths still listed." |
|
||||
| \`coverage_pct_value\` missing, not a number, or outside 0..100 | \`coverage_pct\` | ${degradedMessage(ctx, 'valueCoverageUnavailable')} |
|
||||
| \`coverage_pct_value\` > \`coverage_pct\` | \`coverage_pct\` (clamped) | ${degradedMessage(ctx, 'inconsistentCoverage')} |
|
||||
| Otherwise | \`coverage_pct_value\` | — |
|
||||
|
||||
Y is \`coverage_pct\`; W is \`weak_gaps.length\`; N is \`gaps\`. Remaining slots = 2 × generation cap − tests added or extended so far, and 0 once both passes are used. Option A reads "A) Strengthen the existing ★ test for each weak path and generate tests for true gaps ({slots} of {2 × cap} generation slots remaining)"; at 0 slots it reads "A) List the remaining gaps as proposed tests in the PR body" and dispatches nothing.
|
||||
|
||||
- **>= target:** Pass. "Coverage gate: PASS ({X}% value-weighted)." Continue; list weak paths in the PR body as proposed strengthening.
|
||||
- **>= minimum, < target:** Use AskUserQuestion:
|
||||
- "AI-assessed coverage is {X}%. {N} code paths are untested. Target is {target}%."
|
||||
- RECOMMENDATION: Choose A because untested code paths are where production bugs hide.
|
||||
- "Value-weighted coverage is {X}% ({Y}% including {W} weakly covered paths). {W} paths have only weak tests and {N} have none. Target is {target}%."
|
||||
- RECOMMENDATION: Choose A because weakly covered and untested paths are where regressions slip through.
|
||||
- Options:
|
||||
A) Generate more tests for remaining gaps (recommended)
|
||||
A) (as above, recommended)
|
||||
B) Ship anyway — I accept the coverage risk
|
||||
C) These paths don't need tests — mark as intentionally uncovered
|
||||
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B/C or stop; never another generation pass.
|
||||
C) These paths don't need tests — mark as intentionally uncovered. ${SWEEP_POINTER}
|
||||
- If A and allowance remains: dispatch one generation pass with the weak paths and gaps, then re-evaluate here. At the cap, offer only B/C or stop, plus A as the proposals list; never another generation pass.
|
||||
- If B: Continue. Include in PR body: "Coverage gate: {X}% — user accepted risk."
|
||||
- If C: Continue. Include in PR body: "Coverage gate: {X}% — {N} paths intentionally uncovered."
|
||||
|
||||
- **< minimum:** Use AskUserQuestion:
|
||||
- "AI-assessed coverage is critically low ({X}%). {N} of {M} code paths have no tests. Minimum threshold is {minimum}%."
|
||||
- RECOMMENDATION: Choose A because less than {minimum}% means more code is untested than tested.
|
||||
- "Value-weighted coverage is critically low ({X}%; {Y}% including {W} weakly covered paths). {N} of {M} code paths have no tests. Minimum threshold is {minimum}%."
|
||||
- RECOMMENDATION: Choose A because less than {minimum}% means more behavior is unprotected than protected.
|
||||
- Options:
|
||||
A) Generate tests for remaining gaps (recommended)
|
||||
A) (as above, recommended)
|
||||
B) Override — ship with low coverage (I understand the risk)
|
||||
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B or stop; never another generation pass.
|
||||
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B or stop, plus A as the proposals list; never another generation pass.
|
||||
- If B: Continue. Include in PR body: "Coverage gate: OVERRIDDEN at {X}%."
|
||||
|
||||
**Coverage percentage undetermined:** If the coverage diagram doesn't produce a clear numeric percentage (ambiguous output, parse error), **skip the gate** with: "Coverage gate: could not determine percentage — skipping." Do not default to 0% or block.
|
||||
|
||||
**Test-only diffs:** Skip the gate (same as the existing fast-path).
|
||||
**Spawned or non-interactive session** (the preamble echoed \`SESSION_KIND: spawned\` or \`headless\`): ask nothing. Take A restricted to true \`gaps\` within the remaining slots; never edit tests for weak paths there. List weak paths in the PR body as proposed strengthening.
|
||||
|
||||
**100% coverage:** "Coverage gate: PASS (100%)." Continue.`;
|
||||
|
||||
@@ -590,51 +639,19 @@ Repo: {owner/repo}
|
||||
## Critical Paths
|
||||
- {end-to-end flow that must work}
|
||||
\`\`\``);
|
||||
} else {
|
||||
// review mode
|
||||
sections.push(`
|
||||
**Step 5. Generate tests for gaps (Fix-First):**
|
||||
|
||||
If test framework is detected and gaps were identified:
|
||||
- Classify each gap as AUTO-FIX or ASK per the Fix-First Heuristic:
|
||||
- **AUTO-FIX:** Simple unit tests for pure functions, edge cases of existing tested functions
|
||||
- **ASK:** E2E tests, tests requiring new test infrastructure, tests for ambiguous behavior
|
||||
- For AUTO-FIX gaps: generate the test, run it, commit as \`test: coverage for {feature}\`
|
||||
- For ASK gaps: include in the Fix-First batch question with the other review findings
|
||||
- For paths marked [→E2E]: always ASK (E2E tests are higher-effort and need user confirmation)
|
||||
- For paths marked [→EVAL]: always ASK (eval tests need user confirmation on quality criteria)
|
||||
|
||||
If no test framework detected → include gaps as INFORMATIONAL findings only, no generation.
|
||||
|
||||
**Diff is test-only changes:** Skip Step 4.75 entirely: "No new application code paths to audit."
|
||||
|
||||
### Coverage Warning
|
||||
|
||||
After producing the coverage diagram, check the coverage percentage. Read CLAUDE.md for a \`## Test Coverage\` section with a \`Minimum:\` field. If not found, use default: 60%.
|
||||
|
||||
If coverage is below the minimum threshold, output a prominent warning **before** the regular review findings:
|
||||
|
||||
\`\`\`
|
||||
⚠️ COVERAGE WARNING: AI-assessed coverage is {X}%. {N} code paths untested.
|
||||
Consider writing tests before running /ship.
|
||||
\`\`\`
|
||||
|
||||
This is INFORMATIONAL — does not block /review. But it makes low coverage visible early so the developer can address it before reaching the /ship coverage gate.
|
||||
|
||||
If coverage percentage cannot be determined, skip the warning silently.`);
|
||||
}
|
||||
|
||||
return part === 'gate' ? gate : sections.join('\n');
|
||||
}
|
||||
|
||||
export function generateTestCoverageAuditPlan(_ctx: TemplateContext): string {
|
||||
return generateTestCoverageAuditInner('plan');
|
||||
export function generateTestCoverageAuditPlan(ctx: TemplateContext): string {
|
||||
return generateTestCoverageAuditInner(ctx, 'plan');
|
||||
}
|
||||
|
||||
export function generateTestCoverageAuditShip(_ctx: TemplateContext): string {
|
||||
return generateTestCoverageAuditInner('ship');
|
||||
export function generateTestCoverageAuditShip(ctx: TemplateContext): string {
|
||||
return generateTestCoverageAuditInner(ctx, 'ship');
|
||||
}
|
||||
|
||||
export function generateTestCoverageGateShip(_ctx: TemplateContext): string {
|
||||
return generateTestCoverageAuditInner('ship', 'gate');
|
||||
export function generateTestCoverageGateShip(ctx: TemplateContext): string {
|
||||
return generateTestCoverageAuditInner(ctx, 'ship', 'gate');
|
||||
}
|
||||
Reference in new issue
Block a user