mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
v1.91.9.0 feat: test value bar in plan-eng-review, review, qa and ship, plus /test-audit (#2998)
This commit is contained in:
1 parent
943105f109
commit
96764e80a6
56 files changed
+2445
-216
No files matched your search
@@ -31,8 +31,15 @@ import { skillCensus } from './helpers/skill-census';
|
||||
* New-skill ratchet: previous ceiling 1,150 + ceil(82 / 4) = 1,171
|
||||
* token-equivalents (4,684 bytes), leaving 9 bytes. Existing descriptions
|
||||
* are unchanged. Dominant skill: design-consultation at 229 bytes.
|
||||
* ref test-audit addition on base 943105f1 (2026-09-29)
|
||||
* result pre-addition aggregate 4,680 bytes; test-audit adds 92 bytes
|
||||
* (name + concise description), yielding 4,772 bytes = 1,193
|
||||
* token-equivalents including the root router alias.
|
||||
* New-skill ratchet: previous ceiling 1,171 + ceil(92 / 4) = 1,194
|
||||
* token-equivalents (4,776 bytes), leaving 4 bytes. Existing descriptions
|
||||
* are unchanged.
|
||||
*/
|
||||
const CATALOG_BUDGET_TOKEN_EQUIVALENTS = 1_171;
|
||||
const CATALOG_BUDGET_TOKEN_EQUIVALENTS = 1_194;
|
||||
|
||||
// Largest today: design-consultation at 229 bytes. A description that needs
|
||||
// more than 260 bytes is a body paragraph, not a catalog entry.
|
||||
|
||||
@@ -42,7 +42,7 @@ test('the existing quality and behavior phases retain their complete separate sh
|
||||
expect(quality.evalsAll).toBe(true);
|
||||
expect(behavior.evalsAll).toBe(true);
|
||||
expect(qualityFiles).toHaveLength(1);
|
||||
expect(behaviorFiles).toHaveLength(45);
|
||||
expect(behaviorFiles).toHaveLength(46);
|
||||
expect(behaviorFiles).toEqual(expect.arrayContaining([
|
||||
'test/skill-e2e-qa-callers.test.ts',
|
||||
'test/skill-e2e-qa-functional-fix.test.ts',
|
||||
|
||||
@@ -127,7 +127,7 @@ test('live periodic census fits the declared CI wall including setup', () => {
|
||||
expect(periodicJob['timeout-minutes']).toBe(360);
|
||||
expect(periodicJob.strategy['max-parallel']).toBe(8);
|
||||
expect(Math.max(...walls) + 20 * 60_000).toBeLessThanOrEqual(periodicJob['timeout-minutes'] * 60_000);
|
||||
expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(70);
|
||||
expect(m.entries.filter(e => e.status === 'planned')).toHaveLength(71);
|
||||
const overlays = m.entries.filter(e => e.status === 'planned' && e.slice === periodicSliceCount);
|
||||
expect(overlays).toHaveLength(4);
|
||||
expect(overlays.every(e => isOverlayTestFile(e.file))).toBe(true);
|
||||
@@ -135,7 +135,7 @@ test('live periodic census fits the declared CI wall including setup', () => {
|
||||
|
||||
test('registered allocation is deterministic and preserves every discovered file', () => {
|
||||
const files = collectPaidTestFiles();
|
||||
expect(files).toHaveLength(104);
|
||||
expect(files).toHaveLength(105);
|
||||
expect(files).toContain('test/skill-e2e-ship-skip.test.ts');
|
||||
const m = livePlan(files);
|
||||
expect(livePlan([...files].reverse())).toEqual(m);
|
||||
|
||||
Vendored
+2
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"_comment": "Context-budget ratchet ceilings (~tokens). Regenerate: bun test/helpers/capture-context-budget.ts. Headroom: alwaysOnTotal x1.05, eagerPerInvocation x1.1. Graded by test/context-budget-ratchet.test.ts via lib/context-bill.ts checkBudget.",
|
||||
"alwaysOnTotal": 6465,
|
||||
"alwaysOnTotal": 6873,
|
||||
"eagerPerInvocation": {
|
||||
"autoplan": 18022,
|
||||
"benchmark": 7657,
|
||||
@@ -61,6 +61,7 @@
|
||||
"skillify": 12196,
|
||||
"spec": 14993,
|
||||
"sync-gbrain": 13975,
|
||||
"test-audit": 10439,
|
||||
"unfreeze": 393
|
||||
}
|
||||
}
|
||||
+4
-1
@@ -1152,11 +1152,14 @@ Log metrics for `/retro` through `gstack-review-log`; it handles project/branch
|
||||
JSON validation, storage and sync. It takes **no path argument**; do not build one.
|
||||
|
||||
```bash
|
||||
~/.claude/skills/gstack/bin/gstack-review-log '{"skill":"ship","timestamp":"'"$(date -u +%Y-%m-%dT%H:%M:%SZ)"'","coverage_pct":COVERAGE_PCT,"plan_items_total":PLAN_TOTAL,"plan_items_done":PLAN_DONE,"verification_result":"VERIFY_RESULT","version":"VERSION","branch":"'"$(git rev-parse --abbrev-ref HEAD)"'"}'
|
||||
~/.claude/skills/gstack/bin/gstack-review-log '{"skill":"ship","timestamp":"'"$(date -u +%Y-%m-%dT%H:%M:%SZ)"'","coverage_pct":COVERAGE_PCT,"coverage_schema":2,"coverage_pct_value":COVERAGE_PCT_VALUE,"weak_gaps":WEAK_GAPS,"tests_extended":TESTS_EXTENDED,"tests_rejected":TESTS_REJECTED,"regression_proof":REGRESSION_PROOF,"plan_items_total":PLAN_TOTAL,"plan_items_done":PLAN_DONE,"verification_result":"VERIFY_RESULT","version":"VERSION","branch":"'"$(git rev-parse --abbrev-ref HEAD)"'"}'
|
||||
```
|
||||
|
||||
Substitute from earlier steps:
|
||||
- **COVERAGE_PCT**: Step 7 diagram's integer percentage; encode null/undetermined as -1
|
||||
- **COVERAGE_PCT_VALUE**: Step 7's `coverage_pct_value` (the gate's X) as an integer, or `null` when missing or ignored
|
||||
- **WEAK_GAPS**, **TESTS_EXTENDED**, **TESTS_REJECTED**: counts of Step 7's `weak_gaps`, `tests_extended` and `tests_rejected` (0 when the key is missing or ignored)
|
||||
- **REGRESSION_PROOF**: `{"red_at_head":N,"base_green":N,"base_unavailable":N}` from Step 7, or `null` when missing
|
||||
- **PLAN_TOTAL**: total plan items extracted in Step 8 (0 if no plan file)
|
||||
- **PLAN_DONE**: count of DONE + CHANGED items from Step 8 (0 if no plan file)
|
||||
- **VERIFY_RESULT**: "pass", "fail", or "skipped", set after Step 9 executes Step 8.1's verification list
|
||||
|
||||
+144
-29
@@ -1147,16 +1147,24 @@ the initial audit, failures and zero-test results. Re-entry never resets it.
|
||||
Two passes already used means no further generation; read-only reassessment uses no pass.
|
||||
|
||||
**Subagent prompt:** Supply `<base>`, Step 4's framework/bootstrap decision,
|
||||
permitted paths/commands, remaining gaps, passes used and generation allowance.
|
||||
No allowance means audit only; missing permission is not approval. Preserve the
|
||||
30-path/20-test/2-minute per-test caps.
|
||||
permitted paths/commands, remaining gaps, passes used and generation allowance,
|
||||
plus the AGENTS.md `## Test Coverage` values the gate below reads (`Generation cap:`,
|
||||
`Base control:`, `Base control budget:`). No allowance means audit only; missing
|
||||
permission is not approval. Preserve the 30-path/5-tests-per-pass/2-minute per-test caps.
|
||||
|
||||
**Before the first dispatch,** sweep base-control worktrees a previous interrupted run left behind:
|
||||
|
||||
```bash
|
||||
git worktree prune
|
||||
find "${TMPDIR:-/tmp}" -maxdepth 1 -name 'gstack-base-control.*' -mmin +10 2>/dev/null | while IFS= read -r d; do git worktree remove --force "$d/wt" >/dev/null 2>&1; rm -rf "$d"; done
|
||||
```
|
||||
|
||||
````text
|
||||
You are running a ship-workflow test coverage audit. Run `git diff origin/<base>` to include uncommitted tracked changes; also read relevant non-ignored untracked source/tests. Do not commit or push. Perform only this audit; return unresolved user decisions to the parent instead of asking or advancing to another workflow step.
|
||||
|
||||
Generation: <allowed|audit-only>; passes used: <N> of 2. Audit-only overrides every generation instruction below.
|
||||
|
||||
100% coverage is the goal — every untested path is a path where bugs hide and vibe coding becomes yolo coding. Evaluate what was ACTUALLY coded (from the diff), not what was planned.
|
||||
Coverage goal: every changed behavior is protected by a test that would catch a real regression. Test count is not a goal. Evaluate what was ACTUALLY coded (from the diff), not what was planned.
|
||||
|
||||
### Test Framework Detection
|
||||
|
||||
@@ -1254,7 +1262,27 @@ Go through your diagram branch by branch — both code paths AND user flows. For
|
||||
Quality scoring rubric:
|
||||
- ★★★ Tests behavior with edge cases AND error paths
|
||||
- ★★ Tests correct behavior, happy path only
|
||||
- ★ Smoke test / existence check / trivial assertion (e.g., "it renders", "it doesn't throw")
|
||||
- ★ Smoke test / existence check / trivial assertion (e.g., "it renders", "it doesn't throw"); weak, never counts as coverage
|
||||
|
||||
**Test value bar.** Propose or write a test only with all four answers; otherwise extend an existing test or drop it:
|
||||
|
||||
1. What observable behavior, invariant or independent contract does it protect?
|
||||
2. What credible regression makes it fail?
|
||||
3. Why does existing coverage not already catch that? Prefer adding a row to an existing table-driven test or shared fixture over a near-duplicate.
|
||||
4. Does it need a production seam (export, flag, wrapper, injection hook) that no production caller needs? If yes, test at the real boundary instead.
|
||||
|
||||
A test that breaks under a behavior-preserving refactor asserts implementation: rewrite it at the owning boundary, unless exact output is the declared contract (goldens, prompt bytes, wire formats).
|
||||
|
||||
Value card: `Value: protects=<...>; fails_when=<...>; why_new=<...>; seam=none` (seam: `none` or its name); each field at most 160 UTF-8 bytes here (clamp to 157 plus `...`; JSON keeps full values). Write it as a header comment in each generated test, next to the attribution (wrap, do not truncate); with no known comment syntax, put it in the PR body's Test value details. A missing upstream card never blocks: derive it; ignore unknown fields.
|
||||
|
||||
Example: Value: protects=refundPayment rejects an empty reason; fails_when=the reason guard is removed or inverted; why_new=billing.test.ts covers processPayment only; seam=none
|
||||
Rejected (covered_elsewhere): "checkout renders"; checkout.e2e.ts:15 covers it, so extend that test.
|
||||
|
||||
Weak tests (★ smoke/existence/trivial, gate-failing or unrated) never count as coverage. X = paths with a ★★/★★★ test / total paths (value-weighted; the gate uses X); Y = paths with any test / total paths. Total paths = the diff's codepath trace, max 30; zero skips the gate. A path with only weak tests is uncovered in X, covered in Y, and goes to `weak_gaps` (reason `star_one|gate_failed|unrated`), not `gaps`. Rate stars only for tests reachable from changed paths.
|
||||
|
||||
Retention bar: keep a test that independently enforces a public API, protocol, config, migration, storage, security, platform, default, prompt-byte, generated-output (golden), package, release or architecture contract; static or slow is no reason to delete.
|
||||
|
||||
Regression proof: a regression test must fail at HEAD before any repair, in its own assertion (a pass at HEAD drops the regression label; an import, fixture or env failure is a test defect: correct once or drop). It must pass at base as the control (an assertion failure there marks it invalid; any other failure is "base control unavailable: collection error") and pass after the repair. Record: `Regression proof — fails at HEAD: yes · passes at base: yes | unavailable (<reason>) | manual · passes after fix: yes | pending`.
|
||||
|
||||
### E2E Test Decision Matrix
|
||||
|
||||
@@ -1286,6 +1314,35 @@ A regression is when:
|
||||
|
||||
When uncertain whether a change is a regression, err on the side of writing the test.
|
||||
|
||||
**Red-first proof.** Apply the value bar's Regression proof to every regression test: the diff at HEAD is the pre-fix code, so run the new test at HEAD before any repair. Then, unless the parent says `Base control: off`, run this base control once per regression test in diff order, within a 3-minute total per /ship run (`Base control budget:` seconds per test, default 90). Past the total, record "base control unavailable: budget" and report "N of M regression tests got base control". The block is one shell invocation; it installs nothing and runs no build or postinstall.
|
||||
|
||||
```bash
|
||||
# Set: BASE = the base branch this /ship run resolved; TEST = the test file; FIXTURES = new
|
||||
# test-only fixtures it imports (repo-relative, may be empty); RUN = the detected
|
||||
# runner for one file (e.g. "bun test $TEST"); BUDGET = seconds for this run (default 90).
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
CTL_TMP=$(mktemp -d "${TMPDIR:-/tmp}/gstack-base-control.XXXXXX")
|
||||
cleanup() { git -C "$ROOT" worktree remove --force "$CTL_TMP/wt" >/dev/null 2>&1; rm -rf "$CTL_TMP"; [ -e "$CTL_TMP" ] && echo "BASE_CONTROL_LEFTOVER: $CTL_TMP (run: git worktree prune)"; }
|
||||
trap cleanup EXIT INT TERM
|
||||
(
|
||||
[ -f "$ROOT/package.json" ] || { echo "BASE_CONTROL: unavailable (ecosystem)"; exit 0; }
|
||||
git -C "$ROOT" remote get-url origin >/dev/null 2>&1 || { echo "BASE_CONTROL: unavailable (no base remote)"; exit 0; }
|
||||
timeout 30 git -C "$ROOT" fetch --quiet origin "$BASE" || { echo "BASE_CONTROL: unavailable (base not fetched)"; exit 0; }
|
||||
git -C "$ROOT" worktree add --quiet --detach "$CTL_TMP/wt" "origin/$BASE" >/dev/null 2>&1 || { echo "BASE_CONTROL: unavailable (worktree add failed)"; exit 0; }
|
||||
for f in $TEST $FIXTURES; do mkdir -p "$CTL_TMP/wt/$(dirname "$f")" && cp "$ROOT/$f" "$CTL_TMP/wt/$f"; done
|
||||
[ -d "$ROOT/node_modules" ] && ln -s "$ROOT/node_modules" "$CTL_TMP/wt/node_modules"
|
||||
cd "$CTL_TMP/wt" && timeout "${BUDGET:-90}" sh -c "$RUN" > "$CTL_TMP/out" 2>&1; rc=$?
|
||||
tail -n 40 "$CTL_TMP/out"
|
||||
if [ "$rc" -eq 0 ]; then echo "BASE_CONTROL: passes at base"
|
||||
elif [ "$rc" -eq 124 ]; then echo "BASE_CONTROL: unavailable (budget)"
|
||||
else echo "BASE_CONTROL: fails at base (exit $rc)"; fi
|
||||
)
|
||||
```
|
||||
|
||||
Classify "fails at base" from the output: a failure in the test's own assertion marks it invalid (correct once or drop it); an import, collection, missing generated artifact or dependency failure is "base control unavailable: collection error" and the test stays. Print each unavailable result as: base control unavailable: <reason>. The fails-at-HEAD result still stands. To check by hand: `git worktree add --detach <tmp> <base>`; copy the test and its new fixtures to the same paths; run the detected test command in <tmp>; `git worktree remove --force <tmp>`. Then report `passes at base: manual`. (see $GSTACK_ROOT/docs/test-value-bar.md#base-control-unavailable)
|
||||
|
||||
Return `"regression_proof":{"red_at_head":N,"base_green":N,"base_unavailable":N}` counts in the JSON and each test's record line in the diagram.
|
||||
|
||||
**4. Output ASCII coverage diagram:**
|
||||
|
||||
For targeted audits, start Test review output with the coverage diagram. In full
|
||||
@@ -1323,6 +1380,9 @@ Avoid bare `[ ]` or `[x]` in diagrams unless the block includes
|
||||
**5. Generate tests for uncovered paths:**
|
||||
|
||||
If test framework detected (or bootstrapped in Step 4):
|
||||
- Apply the test value bar before writing each test. Extend an existing test (a new table row, fixture case or assertion) before creating a file. Record every proposal you decline in `tests_rejected` with a `reason_code` from `duplicate_protects, needs_seam, incomplete_card, no_credible_regression, covered_elsewhere, implementation_coupled`.
|
||||
- Never add a production seam for a test; a seam that is not `none` names its non-test callers: `seam=<name> (non-test callers: N, via <search command>)`.
|
||||
- Write the value card as a header comment in each generated or extended test.
|
||||
- Prioritize error handlers and edge cases first (happy paths are more likely already tested)
|
||||
- Read 2-3 existing test files to match conventions exactly
|
||||
- Generate unit tests. Mock all external dependencies (DB, API, Redis).
|
||||
@@ -1332,7 +1392,9 @@ If test framework detected (or bootstrapped in Step 4):
|
||||
- Run each test. Passes → keep the change and report its path; the parent commits in Step 15.
|
||||
- Fails → diagnose whether the test/fixture is invalid or a declared product contract is broken. Correct a demonstrated test defect once; preserve a valid red regression and route the reproduced product failure through the parent's fix/approval flow. Never delete or weaken it to manufacture green; retain unresolved coverage in the diagram.
|
||||
|
||||
Caps: 30 code paths max, 20 tests generated max (code + user flow combined), 2-min per-test exploration cap.
|
||||
Caps: 30 code paths max; 5 tests per generation pass (code + user flow combined; the parent's `Generation cap:` overrides 5); an extension uses one slot and a rejection uses none; 2-min per-test exploration cap. List each remaining gap below the diagram (inside `diagram`) as a proposed test with its value card.
|
||||
|
||||
Do not rate stars for tests you wrote in this pass: count them as unrated (weak, reason `unrated`). The parent's read-only rating dispatch rates them. Counts are disjoint, precedence extended > added > rejected: one gap lands in at most one of `tests_extended`, `tests_added`, `tests_rejected`.
|
||||
|
||||
If no test framework AND user declined bootstrap → diagram only, no generation. Note: "Test generation skipped — no test framework configured."
|
||||
|
||||
@@ -1346,7 +1408,7 @@ git ls-files 2>/dev/null | grep -E '(\.test\.|\.spec\.|_test\.|_spec\.)' | wc -l
|
||||
```
|
||||
|
||||
For PR body: `Tests: {before} → {after} (+{delta} new)`
|
||||
Coverage line: `Test Coverage Audit: N new code paths. M covered (X%). K tests generated, awaiting parent commit.`
|
||||
Coverage line: `Test Coverage Audit: N new code paths. M covered (Y% any test, X% value-weighted). K tests generated, awaiting parent commit.`
|
||||
|
||||
### Test Plan Artifact
|
||||
|
||||
@@ -1380,16 +1442,52 @@ Repo: {owner/repo}
|
||||
```
|
||||
|
||||
After your analysis, output a single JSON object on the LAST LINE of your response (no other text after it):
|
||||
{"coverage_pct":N,"gaps":N,"diagram":"<full markdown coverage diagram for PR body>","tests_added":["path",...]}
|
||||
Use null for an undetermined or skipped coverage percentage, not zero. Include every remaining gap in the diagram so the parent can target a second pass.
|
||||
{"coverage_pct":N,"gaps":N,"diagram":"<full markdown coverage diagram for PR body>","tests_added":["path",...],"coverage_pct_value":N,"weak_gaps":[{"path":"...","existing_test":"...","reason":"star_one|gate_failed|unrated"}],"tests_extended":["path",...],"tests_rejected":[{"path_or_gap":"...","reason_code":"...","reason":"..."}],"regression_proof":{"red_at_head":N,"base_green":N,"base_unavailable":N}}
|
||||
`coverage_pct` is Y (paths with any test), `coverage_pct_value` is X (paths with a ★★/★★★ test), `gaps` counts only paths with no test. Use null for an undetermined or skipped coverage percentage, not zero. Include every remaining gap in the diagram so the parent can target a second pass.
|
||||
````
|
||||
|
||||
**Parent processing:**
|
||||
|
||||
1. Read the subagent's final output. Parse the LAST line as JSON.
|
||||
2. Store `coverage_pct` (for Step 20 metrics), `gaps` (user summary), `tests_added` (for the commit).
|
||||
3. Embed `diagram` verbatim in the PR body's `## Test Coverage` section (Step 19).
|
||||
4. Print a one-line summary: `Coverage: {coverage_pct}%, {gaps} gaps. {tests_added.length} tests added.`
|
||||
2. Store `coverage_pct`, `coverage_pct_value`, `gaps`, `weak_gaps`, `tests_added`,
|
||||
`tests_extended`, `tests_rejected` and `regression_proof`. A missing new key counts
|
||||
as empty; say so in the summary (an older installed prompt must not fail the gate).
|
||||
A key with the wrong type (for example `weak_gaps` not an array) is ignored the same
|
||||
way and printed as: malformed <key> ignored: the audit returned the wrong type, so it counts as empty. The likely cause is an outdated installed skill; run /gstack-upgrade. (see $GSTACK_ROOT/docs/test-value-bar.md#malformed-key-ignored)
|
||||
3. **Machine checks** on every test written in this run (`tests_added` and
|
||||
`tests_extended`): a value-card header with four non-empty fields (else
|
||||
`incomplete_card`); `protects` unique across the run after casefolding and stripping
|
||||
punctuation and repeated whitespace (a later duplicate is `duplicate_protects`); seam
|
||||
`none`, or a named seam with at least one non-test caller (N = 0 or an unavailable
|
||||
caller check is `needs_seam`). Move each failure to `tests_rejected` with its
|
||||
`reason_code`, then remove it before anything else reads the diff: an untracked new
|
||||
file is deleted; for a tracked file, revert only this run's hunk with Edit, never
|
||||
the whole file.
|
||||
|
||||
```bash
|
||||
while IFS= read -r f; do
|
||||
[ -n "$f" ] || continue
|
||||
if git ls-files --error-unmatch -- "$f" >/dev/null 2>&1; then echo "REVERT_HUNK: $f"; else rm -f -- "$f" && echo "REMOVED: $f"; fi
|
||||
done <<'REJECTED'
|
||||
<one rejected test path per line>
|
||||
REJECTED
|
||||
```
|
||||
|
||||
No `tests_rejected` path may remain on disk as a new file. If every test written in
|
||||
a pass is rejected, print all <N> generated tests rejected by machine checks; see tests_rejected. The gate proceeds with the unchanged value-weighted coverage. (see $GSTACK_ROOT/docs/test-value-bar.md#all-generated-tests-rejected)
|
||||
4. **Rating dispatch.** When this run wrote tests that survived the machine checks,
|
||||
dispatch one read-only Agent (`subagent_type: "general-purpose"`,
|
||||
`run_in_background: false`) with no generation permission; it uses no generation
|
||||
pass. Give it the diagram and the surviving test paths. It rates each against the
|
||||
★ rubric and the test value bar and returns a LAST-line JSON
|
||||
`{"coverage_pct_value":N,"weak_gaps":[...]}` recomputed with its ratings; use those
|
||||
two values. Until rated, this run's tests count as weak (`unrated`). If it fails,
|
||||
times out or returns invalid JSON, the gate is skipped for this run ("rating
|
||||
unavailable").
|
||||
5. Embed `diagram` verbatim in the PR body's `## Test Coverage` section (Step 19).
|
||||
6. Print a one-line summary: `Coverage: {X}% value-weighted ({Y}% including {W} weakly covered paths), {gaps} gaps. {tests_added.length} tests added.`
|
||||
Bindings for the PR body's Test value line: K = `tests_added.length`,
|
||||
R = `tests_rejected.length`, E = `tests_extended.length`, W = `weak_gaps.length`.
|
||||
|
||||
**Audit failure:** On failure, invalid JSON or no completion after ~10 minutes,
|
||||
stop the child and confirm it stopped before running the same audit inline.
|
||||
@@ -1400,36 +1498,45 @@ and test-only rules. Preserve partial results as incomplete, not passing coverag
|
||||
|
||||
**7. Coverage gate:**
|
||||
|
||||
The parent owns this gate, including after inline fallback. Generated tests stay uncommitted until Step 15. Use Step 7's remaining generation allowance; supply it and the remaining gaps to the same audit prompt. At the cap, omit A and recommend stopping; the listed risk choices remain available.
|
||||
The parent owns this gate, including after inline fallback. Generated tests stay uncommitted until Step 15. The gate only asks; it never hard-fails. Use Step 7's remaining generation allowance; supply it and the remaining gaps to the same audit prompt. At the cap, omit A's generation pass and recommend stopping; A then only lists proposals and the listed risk choices remain available.
|
||||
|
||||
Before proceeding, check AGENTS.md for a `## Test Coverage` section with `Minimum:` and `Target:` fields. If found, use those percentages. Otherwise use defaults: Minimum = 60%, Target = 80%.
|
||||
Read AGENTS.md's `## Test Coverage` section for `Minimum:` and `Target:`; otherwise use defaults: Minimum = 60%, Target = 80%. Also read the optional `Generation cap:` (tests per pass, default 5), `Base control:` (`auto` default, or `off`), `Base control budget:` (seconds per run, default 90) and `Star rating:` (`auto` default, or `off`). Missing keys use the defaults.
|
||||
|
||||
Using the coverage percentage from the diagram in substep 4 (the `COVERAGE: X/Y (Z%)` line):
|
||||
**Gate number X.** Take the first matching row; never substitute 0:
|
||||
|
||||
- **>= target:** Pass. "Coverage gate: PASS ({X}%)." Continue.
|
||||
| Step 7 result | Gate number | Print |
|
||||
|---|---|---|
|
||||
| Rating dispatch failed or timed out | skip the gate | rating unavailable: the read-only rating dispatch failed or timed out, so the coverage gate is skipped for this run. Re-run Step 7 to re-rate the tests. (see $GSTACK_ROOT/docs/test-value-bar.md#rating-unavailable) |
|
||||
| Zero paths, test-only diff, or `coverage_pct` null or unparseable | skip the gate | "Coverage gate: could not determine percentage — skipping." |
|
||||
| `Star rating: off` | `coverage_pct` | "Star rating off: gate uses coverage_pct; weak paths still listed." |
|
||||
| `coverage_pct_value` missing, not a number, or outside 0..100 | `coverage_pct` | value-weighted coverage unavailable (outdated installed skill); run /gstack-upgrade. The gate used coverage_pct (any test) this run. (see $GSTACK_ROOT/docs/test-value-bar.md#value-weighted-coverage-unavailable) |
|
||||
| `coverage_pct_value` > `coverage_pct` | `coverage_pct` (clamped) | inconsistent coverage inputs: coverage_pct_value was above coverage_pct, so it was clamped to coverage_pct. Re-run Step 7 if the numbers look wrong. (see $GSTACK_ROOT/docs/test-value-bar.md#inconsistent-coverage-inputs) |
|
||||
| Otherwise | `coverage_pct_value` | — |
|
||||
|
||||
Y is `coverage_pct`; W is `weak_gaps.length`; N is `gaps`. Remaining slots = 2 × generation cap − tests added or extended so far, and 0 once both passes are used. Option A reads "A) Strengthen the existing ★ test for each weak path and generate tests for true gaps ({slots} of {2 × cap} generation slots remaining)"; at 0 slots it reads "A) List the remaining gaps as proposed tests in the PR body" and dispatches nothing.
|
||||
|
||||
- **>= target:** Pass. "Coverage gate: PASS ({X}% value-weighted)." Continue; list weak paths in the PR body as proposed strengthening.
|
||||
- **>= minimum, < target:** Use AskUserQuestion:
|
||||
- "AI-assessed coverage is {X}%. {N} code paths are untested. Target is {target}%."
|
||||
- RECOMMENDATION: Choose A because untested code paths are where production bugs hide.
|
||||
- "Value-weighted coverage is {X}% ({Y}% including {W} weakly covered paths). {W} paths have only weak tests and {N} have none. Target is {target}%."
|
||||
- RECOMMENDATION: Choose A because weakly covered and untested paths are where regressions slip through.
|
||||
- Options:
|
||||
A) Generate more tests for remaining gaps (recommended)
|
||||
A) (as above, recommended)
|
||||
B) Ship anyway — I accept the coverage risk
|
||||
C) These paths don't need tests — mark as intentionally uncovered
|
||||
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B/C or stop; never another generation pass.
|
||||
C) These paths don't need tests — mark as intentionally uncovered. Repo-wide sweep: run /test-audit.
|
||||
- If A and allowance remains: dispatch one generation pass with the weak paths and gaps, then re-evaluate here. At the cap, offer only B/C or stop, plus A as the proposals list; never another generation pass.
|
||||
- If B: Continue. Include in PR body: "Coverage gate: {X}% — user accepted risk."
|
||||
- If C: Continue. Include in PR body: "Coverage gate: {X}% — {N} paths intentionally uncovered."
|
||||
|
||||
- **< minimum:** Use AskUserQuestion:
|
||||
- "AI-assessed coverage is critically low ({X}%). {N} of {M} code paths have no tests. Minimum threshold is {minimum}%."
|
||||
- RECOMMENDATION: Choose A because less than {minimum}% means more code is untested than tested.
|
||||
- "Value-weighted coverage is critically low ({X}%; {Y}% including {W} weakly covered paths). {N} of {M} code paths have no tests. Minimum threshold is {minimum}%."
|
||||
- RECOMMENDATION: Choose A because less than {minimum}% means more behavior is unprotected than protected.
|
||||
- Options:
|
||||
A) Generate tests for remaining gaps (recommended)
|
||||
A) (as above, recommended)
|
||||
B) Override — ship with low coverage (I understand the risk)
|
||||
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B or stop; never another generation pass.
|
||||
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B or stop, plus A as the proposals list; never another generation pass.
|
||||
- If B: Continue. Include in PR body: "Coverage gate: OVERRIDDEN at {X}%."
|
||||
|
||||
**Coverage percentage undetermined:** If the coverage diagram doesn't produce a clear numeric percentage (ambiguous output, parse error), **skip the gate** with: "Coverage gate: could not determine percentage — skipping." Do not default to 0% or block.
|
||||
|
||||
**Test-only diffs:** Skip the gate (same as the existing fast-path).
|
||||
**Spawned or non-interactive session** (the preamble echoed `SESSION_KIND: spawned` or `headless`): ask nothing. Take A restricted to true `gaps` within the remaining slots; never edit tests for weak paths there. List weak paths in the PR body as proposed strengthening.
|
||||
|
||||
**100% coverage:** "Coverage gate: PASS (100%)." Continue.
|
||||
|
||||
@@ -3183,6 +3290,11 @@ theme, excluding VERSION/CHANGELOG bookkeeping. Do not paste the commit list.>
|
||||
## Test Coverage
|
||||
<coverage diagram from Step 7, or "All new code paths have test coverage.">
|
||||
<If Step 7 ran: "Tests: {before} → {after} (+{delta} new)">
|
||||
<If Step 7 ran: "Coverage: {X}% value-weighted ({Y}% including {W} weakly covered paths)">
|
||||
<If Step 7 ran: "Test value: {K} tests written, {R} rejected by the authoring gate, {E} existing tests extended, {W} paths weakly covered (weak = ★, gate-failing or unrated)." Use the singular noun for a count of 1 ("1 test written", "1 existing test extended", "1 path weakly covered").>
|
||||
<Weak paths and leftover gaps as proposed tests with value cards; each regression test's
|
||||
"Regression proof — fails at HEAD · passes at base · passes after fix" line; Test value
|
||||
details for cards whose file type has no known comment syntax.>
|
||||
|
||||
## Pre-Landing Review
|
||||
<findings from Step 9 code review, or "No issues found.">
|
||||
@@ -3319,11 +3431,14 @@ Log metrics for `/retro` through `gstack-review-log`; it handles project/branch
|
||||
JSON validation, storage and sync. It takes **no path argument**; do not build one.
|
||||
|
||||
```bash
|
||||
$GSTACK_ROOT/bin/gstack-review-log '{"skill":"ship","timestamp":"'"$(date -u +%Y-%m-%dT%H:%M:%SZ)"'","coverage_pct":COVERAGE_PCT,"plan_items_total":PLAN_TOTAL,"plan_items_done":PLAN_DONE,"verification_result":"VERIFY_RESULT","version":"VERSION","branch":"'"$(git rev-parse --abbrev-ref HEAD)"'"}'
|
||||
$GSTACK_ROOT/bin/gstack-review-log '{"skill":"ship","timestamp":"'"$(date -u +%Y-%m-%dT%H:%M:%SZ)"'","coverage_pct":COVERAGE_PCT,"coverage_schema":2,"coverage_pct_value":COVERAGE_PCT_VALUE,"weak_gaps":WEAK_GAPS,"tests_extended":TESTS_EXTENDED,"tests_rejected":TESTS_REJECTED,"regression_proof":REGRESSION_PROOF,"plan_items_total":PLAN_TOTAL,"plan_items_done":PLAN_DONE,"verification_result":"VERIFY_RESULT","version":"VERSION","branch":"'"$(git rev-parse --abbrev-ref HEAD)"'"}'
|
||||
```
|
||||
|
||||
Substitute from earlier steps:
|
||||
- **COVERAGE_PCT**: Step 7 diagram's integer percentage; encode null/undetermined as -1
|
||||
- **COVERAGE_PCT_VALUE**: Step 7's `coverage_pct_value` (the gate's X) as an integer, or `null` when missing or ignored
|
||||
- **WEAK_GAPS**, **TESTS_EXTENDED**, **TESTS_REJECTED**: counts of Step 7's `weak_gaps`, `tests_extended` and `tests_rejected` (0 when the key is missing or ignored)
|
||||
- **REGRESSION_PROOF**: `{"red_at_head":N,"base_green":N,"base_unavailable":N}` from Step 7, or `null` when missing
|
||||
- **PLAN_TOTAL**: total plan items extracted in Step 8 (0 if no plan file)
|
||||
- **PLAN_DONE**: count of DONE + CHANGED items from Step 8 (0 if no plan file)
|
||||
- **VERIFY_RESULT**: "pass", "fail", or "skipped", set after Step 9 executes Step 8.1's verification list
|
||||
|
||||
+144
-29
@@ -1127,16 +1127,24 @@ the initial audit, failures and zero-test results. Re-entry never resets it.
|
||||
Two passes already used means no further generation; read-only reassessment uses no pass.
|
||||
|
||||
**Subagent prompt:** Supply `<base>`, Step 4's framework/bootstrap decision,
|
||||
permitted paths/commands, remaining gaps, passes used and generation allowance.
|
||||
No allowance means audit only; missing permission is not approval. Preserve the
|
||||
30-path/20-test/2-minute per-test caps.
|
||||
permitted paths/commands, remaining gaps, passes used and generation allowance,
|
||||
plus the CLAUDE.md `## Test Coverage` values the gate below reads (`Generation cap:`,
|
||||
`Base control:`, `Base control budget:`). No allowance means audit only; missing
|
||||
permission is not approval. Preserve the 30-path/5-tests-per-pass/2-minute per-test caps.
|
||||
|
||||
**Before the first dispatch,** sweep base-control worktrees a previous interrupted run left behind:
|
||||
|
||||
```bash
|
||||
git worktree prune
|
||||
find "${TMPDIR:-/tmp}" -maxdepth 1 -name 'gstack-base-control.*' -mmin +10 2>/dev/null | while IFS= read -r d; do git worktree remove --force "$d/wt" >/dev/null 2>&1; rm -rf "$d"; done
|
||||
```
|
||||
|
||||
````text
|
||||
You are running a ship-workflow test coverage audit. Run `git diff origin/<base>` to include uncommitted tracked changes; also read relevant non-ignored untracked source/tests. Do not commit or push. Perform only this audit; return unresolved user decisions to the parent instead of asking or advancing to another workflow step.
|
||||
|
||||
Generation: <allowed|audit-only>; passes used: <N> of 2. Audit-only overrides every generation instruction below.
|
||||
|
||||
100% coverage is the goal — every untested path is a path where bugs hide and vibe coding becomes yolo coding. Evaluate what was ACTUALLY coded (from the diff), not what was planned.
|
||||
Coverage goal: every changed behavior is protected by a test that would catch a real regression. Test count is not a goal. Evaluate what was ACTUALLY coded (from the diff), not what was planned.
|
||||
|
||||
### Test Framework Detection
|
||||
|
||||
@@ -1234,7 +1242,27 @@ Go through your diagram branch by branch — both code paths AND user flows. For
|
||||
Quality scoring rubric:
|
||||
- ★★★ Tests behavior with edge cases AND error paths
|
||||
- ★★ Tests correct behavior, happy path only
|
||||
- ★ Smoke test / existence check / trivial assertion (e.g., "it renders", "it doesn't throw")
|
||||
- ★ Smoke test / existence check / trivial assertion (e.g., "it renders", "it doesn't throw"); weak, never counts as coverage
|
||||
|
||||
**Test value bar.** Propose or write a test only with all four answers; otherwise extend an existing test or drop it:
|
||||
|
||||
1. What observable behavior, invariant or independent contract does it protect?
|
||||
2. What credible regression makes it fail?
|
||||
3. Why does existing coverage not already catch that? Prefer adding a row to an existing table-driven test or shared fixture over a near-duplicate.
|
||||
4. Does it need a production seam (export, flag, wrapper, injection hook) that no production caller needs? If yes, test at the real boundary instead.
|
||||
|
||||
A test that breaks under a behavior-preserving refactor asserts implementation: rewrite it at the owning boundary, unless exact output is the declared contract (goldens, prompt bytes, wire formats).
|
||||
|
||||
Value card: `Value: protects=<...>; fails_when=<...>; why_new=<...>; seam=none` (seam: `none` or its name); each field at most 160 UTF-8 bytes here (clamp to 157 plus `...`; JSON keeps full values). Write it as a header comment in each generated test, next to the attribution (wrap, do not truncate); with no known comment syntax, put it in the PR body's Test value details. A missing upstream card never blocks: derive it; ignore unknown fields.
|
||||
|
||||
Example: Value: protects=refundPayment rejects an empty reason; fails_when=the reason guard is removed or inverted; why_new=billing.test.ts covers processPayment only; seam=none
|
||||
Rejected (covered_elsewhere): "checkout renders"; checkout.e2e.ts:15 covers it, so extend that test.
|
||||
|
||||
Weak tests (★ smoke/existence/trivial, gate-failing or unrated) never count as coverage. X = paths with a ★★/★★★ test / total paths (value-weighted; the gate uses X); Y = paths with any test / total paths. Total paths = the diff's codepath trace, max 30; zero skips the gate. A path with only weak tests is uncovered in X, covered in Y, and goes to `weak_gaps` (reason `star_one|gate_failed|unrated`), not `gaps`. Rate stars only for tests reachable from changed paths.
|
||||
|
||||
Retention bar: keep a test that independently enforces a public API, protocol, config, migration, storage, security, platform, default, prompt-byte, generated-output (golden), package, release or architecture contract; static or slow is no reason to delete.
|
||||
|
||||
Regression proof: a regression test must fail at HEAD before any repair, in its own assertion (a pass at HEAD drops the regression label; an import, fixture or env failure is a test defect: correct once or drop). It must pass at base as the control (an assertion failure there marks it invalid; any other failure is "base control unavailable: collection error") and pass after the repair. Record: `Regression proof — fails at HEAD: yes · passes at base: yes | unavailable (<reason>) | manual · passes after fix: yes | pending`.
|
||||
|
||||
### E2E Test Decision Matrix
|
||||
|
||||
@@ -1266,6 +1294,35 @@ A regression is when:
|
||||
|
||||
When uncertain whether a change is a regression, err on the side of writing the test.
|
||||
|
||||
**Red-first proof.** Apply the value bar's Regression proof to every regression test: the diff at HEAD is the pre-fix code, so run the new test at HEAD before any repair. Then, unless the parent says `Base control: off`, run this base control once per regression test in diff order, within a 3-minute total per /ship run (`Base control budget:` seconds per test, default 90). Past the total, record "base control unavailable: budget" and report "N of M regression tests got base control". The block is one shell invocation; it installs nothing and runs no build or postinstall.
|
||||
|
||||
```bash
|
||||
# Set: BASE = the base branch this /ship run resolved; TEST = the test file; FIXTURES = new
|
||||
# test-only fixtures it imports (repo-relative, may be empty); RUN = the detected
|
||||
# runner for one file (e.g. "bun test $TEST"); BUDGET = seconds for this run (default 90).
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
CTL_TMP=$(mktemp -d "${TMPDIR:-/tmp}/gstack-base-control.XXXXXX")
|
||||
cleanup() { git -C "$ROOT" worktree remove --force "$CTL_TMP/wt" >/dev/null 2>&1; rm -rf "$CTL_TMP"; [ -e "$CTL_TMP" ] && echo "BASE_CONTROL_LEFTOVER: $CTL_TMP (run: git worktree prune)"; }
|
||||
trap cleanup EXIT INT TERM
|
||||
(
|
||||
[ -f "$ROOT/package.json" ] || { echo "BASE_CONTROL: unavailable (ecosystem)"; exit 0; }
|
||||
git -C "$ROOT" remote get-url origin >/dev/null 2>&1 || { echo "BASE_CONTROL: unavailable (no base remote)"; exit 0; }
|
||||
timeout 30 git -C "$ROOT" fetch --quiet origin "$BASE" || { echo "BASE_CONTROL: unavailable (base not fetched)"; exit 0; }
|
||||
git -C "$ROOT" worktree add --quiet --detach "$CTL_TMP/wt" "origin/$BASE" >/dev/null 2>&1 || { echo "BASE_CONTROL: unavailable (worktree add failed)"; exit 0; }
|
||||
for f in $TEST $FIXTURES; do mkdir -p "$CTL_TMP/wt/$(dirname "$f")" && cp "$ROOT/$f" "$CTL_TMP/wt/$f"; done
|
||||
[ -d "$ROOT/node_modules" ] && ln -s "$ROOT/node_modules" "$CTL_TMP/wt/node_modules"
|
||||
cd "$CTL_TMP/wt" && timeout "${BUDGET:-90}" sh -c "$RUN" > "$CTL_TMP/out" 2>&1; rc=$?
|
||||
tail -n 40 "$CTL_TMP/out"
|
||||
if [ "$rc" -eq 0 ]; then echo "BASE_CONTROL: passes at base"
|
||||
elif [ "$rc" -eq 124 ]; then echo "BASE_CONTROL: unavailable (budget)"
|
||||
else echo "BASE_CONTROL: fails at base (exit $rc)"; fi
|
||||
)
|
||||
```
|
||||
|
||||
Classify "fails at base" from the output: a failure in the test's own assertion marks it invalid (correct once or drop it); an import, collection, missing generated artifact or dependency failure is "base control unavailable: collection error" and the test stays. Print each unavailable result as: base control unavailable: <reason>. The fails-at-HEAD result still stands. To check by hand: `git worktree add --detach <tmp> <base>`; copy the test and its new fixtures to the same paths; run the detected test command in <tmp>; `git worktree remove --force <tmp>`. Then report `passes at base: manual`. (see $GSTACK_ROOT/docs/test-value-bar.md#base-control-unavailable)
|
||||
|
||||
Return `"regression_proof":{"red_at_head":N,"base_green":N,"base_unavailable":N}` counts in the JSON and each test's record line in the diagram.
|
||||
|
||||
**4. Output ASCII coverage diagram:**
|
||||
|
||||
For targeted audits, start Test review output with the coverage diagram. In full
|
||||
@@ -1303,6 +1360,9 @@ Avoid bare `[ ]` or `[x]` in diagrams unless the block includes
|
||||
**5. Generate tests for uncovered paths:**
|
||||
|
||||
If test framework detected (or bootstrapped in Step 4):
|
||||
- Apply the test value bar before writing each test. Extend an existing test (a new table row, fixture case or assertion) before creating a file. Record every proposal you decline in `tests_rejected` with a `reason_code` from `duplicate_protects, needs_seam, incomplete_card, no_credible_regression, covered_elsewhere, implementation_coupled`.
|
||||
- Never add a production seam for a test; a seam that is not `none` names its non-test callers: `seam=<name> (non-test callers: N, via <search command>)`.
|
||||
- Write the value card as a header comment in each generated or extended test.
|
||||
- Prioritize error handlers and edge cases first (happy paths are more likely already tested)
|
||||
- Read 2-3 existing test files to match conventions exactly
|
||||
- Generate unit tests. Mock all external dependencies (DB, API, Redis).
|
||||
@@ -1312,7 +1372,9 @@ If test framework detected (or bootstrapped in Step 4):
|
||||
- Run each test. Passes → keep the change and report its path; the parent commits in Step 15.
|
||||
- Fails → diagnose whether the test/fixture is invalid or a declared product contract is broken. Correct a demonstrated test defect once; preserve a valid red regression and route the reproduced product failure through the parent's fix/approval flow. Never delete or weaken it to manufacture green; retain unresolved coverage in the diagram.
|
||||
|
||||
Caps: 30 code paths max, 20 tests generated max (code + user flow combined), 2-min per-test exploration cap.
|
||||
Caps: 30 code paths max; 5 tests per generation pass (code + user flow combined; the parent's `Generation cap:` overrides 5); an extension uses one slot and a rejection uses none; 2-min per-test exploration cap. List each remaining gap below the diagram (inside `diagram`) as a proposed test with its value card.
|
||||
|
||||
Do not rate stars for tests you wrote in this pass: count them as unrated (weak, reason `unrated`). The parent's read-only rating dispatch rates them. Counts are disjoint, precedence extended > added > rejected: one gap lands in at most one of `tests_extended`, `tests_added`, `tests_rejected`.
|
||||
|
||||
If no test framework AND user declined bootstrap → diagram only, no generation. Note: "Test generation skipped — no test framework configured."
|
||||
|
||||
@@ -1326,7 +1388,7 @@ git ls-files 2>/dev/null | grep -E '(\.test\.|\.spec\.|_test\.|_spec\.)' | wc -l
|
||||
```
|
||||
|
||||
For PR body: `Tests: {before} → {after} (+{delta} new)`
|
||||
Coverage line: `Test Coverage Audit: N new code paths. M covered (X%). K tests generated, awaiting parent commit.`
|
||||
Coverage line: `Test Coverage Audit: N new code paths. M covered (Y% any test, X% value-weighted). K tests generated, awaiting parent commit.`
|
||||
|
||||
### Test Plan Artifact
|
||||
|
||||
@@ -1360,16 +1422,52 @@ Repo: {owner/repo}
|
||||
```
|
||||
|
||||
After your analysis, output a single JSON object on the LAST LINE of your response (no other text after it):
|
||||
{"coverage_pct":N,"gaps":N,"diagram":"<full markdown coverage diagram for PR body>","tests_added":["path",...]}
|
||||
Use null for an undetermined or skipped coverage percentage, not zero. Include every remaining gap in the diagram so the parent can target a second pass.
|
||||
{"coverage_pct":N,"gaps":N,"diagram":"<full markdown coverage diagram for PR body>","tests_added":["path",...],"coverage_pct_value":N,"weak_gaps":[{"path":"...","existing_test":"...","reason":"star_one|gate_failed|unrated"}],"tests_extended":["path",...],"tests_rejected":[{"path_or_gap":"...","reason_code":"...","reason":"..."}],"regression_proof":{"red_at_head":N,"base_green":N,"base_unavailable":N}}
|
||||
`coverage_pct` is Y (paths with any test), `coverage_pct_value` is X (paths with a ★★/★★★ test), `gaps` counts only paths with no test. Use null for an undetermined or skipped coverage percentage, not zero. Include every remaining gap in the diagram so the parent can target a second pass.
|
||||
````
|
||||
|
||||
**Parent processing:**
|
||||
|
||||
1. Read the subagent's final output. Parse the LAST line as JSON.
|
||||
2. Store `coverage_pct` (for Step 20 metrics), `gaps` (user summary), `tests_added` (for the commit).
|
||||
3. Embed `diagram` verbatim in the PR body's `## Test Coverage` section (Step 19).
|
||||
4. Print a one-line summary: `Coverage: {coverage_pct}%, {gaps} gaps. {tests_added.length} tests added.`
|
||||
2. Store `coverage_pct`, `coverage_pct_value`, `gaps`, `weak_gaps`, `tests_added`,
|
||||
`tests_extended`, `tests_rejected` and `regression_proof`. A missing new key counts
|
||||
as empty; say so in the summary (an older installed prompt must not fail the gate).
|
||||
A key with the wrong type (for example `weak_gaps` not an array) is ignored the same
|
||||
way and printed as: malformed <key> ignored: the audit returned the wrong type, so it counts as empty. The likely cause is an outdated installed skill; run /gstack-upgrade. (see $GSTACK_ROOT/docs/test-value-bar.md#malformed-key-ignored)
|
||||
3. **Machine checks** on every test written in this run (`tests_added` and
|
||||
`tests_extended`): a value-card header with four non-empty fields (else
|
||||
`incomplete_card`); `protects` unique across the run after casefolding and stripping
|
||||
punctuation and repeated whitespace (a later duplicate is `duplicate_protects`); seam
|
||||
`none`, or a named seam with at least one non-test caller (N = 0 or an unavailable
|
||||
caller check is `needs_seam`). Move each failure to `tests_rejected` with its
|
||||
`reason_code`, then remove it before anything else reads the diff: an untracked new
|
||||
file is deleted; for a tracked file, revert only this run's hunk with Edit, never
|
||||
the whole file.
|
||||
|
||||
```bash
|
||||
while IFS= read -r f; do
|
||||
[ -n "$f" ] || continue
|
||||
if git ls-files --error-unmatch -- "$f" >/dev/null 2>&1; then echo "REVERT_HUNK: $f"; else rm -f -- "$f" && echo "REMOVED: $f"; fi
|
||||
done <<'REJECTED'
|
||||
<one rejected test path per line>
|
||||
REJECTED
|
||||
```
|
||||
|
||||
No `tests_rejected` path may remain on disk as a new file. If every test written in
|
||||
a pass is rejected, print all <N> generated tests rejected by machine checks; see tests_rejected. The gate proceeds with the unchanged value-weighted coverage. (see $GSTACK_ROOT/docs/test-value-bar.md#all-generated-tests-rejected)
|
||||
4. **Rating dispatch.** When this run wrote tests that survived the machine checks,
|
||||
dispatch one read-only Agent (`subagent_type: "general-purpose"`,
|
||||
`run_in_background: false`) with no generation permission; it uses no generation
|
||||
pass. Give it the diagram and the surviving test paths. It rates each against the
|
||||
★ rubric and the test value bar and returns a LAST-line JSON
|
||||
`{"coverage_pct_value":N,"weak_gaps":[...]}` recomputed with its ratings; use those
|
||||
two values. Until rated, this run's tests count as weak (`unrated`). If it fails,
|
||||
times out or returns invalid JSON, the gate is skipped for this run ("rating
|
||||
unavailable").
|
||||
5. Embed `diagram` verbatim in the PR body's `## Test Coverage` section (Step 19).
|
||||
6. Print a one-line summary: `Coverage: {X}% value-weighted ({Y}% including {W} weakly covered paths), {gaps} gaps. {tests_added.length} tests added.`
|
||||
Bindings for the PR body's Test value line: K = `tests_added.length`,
|
||||
R = `tests_rejected.length`, E = `tests_extended.length`, W = `weak_gaps.length`.
|
||||
|
||||
**Audit failure:** On failure, invalid JSON or no completion after ~10 minutes,
|
||||
stop the child and confirm it stopped before running the same audit inline.
|
||||
@@ -1380,36 +1478,45 @@ and test-only rules. Preserve partial results as incomplete, not passing coverag
|
||||
|
||||
**7. Coverage gate:**
|
||||
|
||||
The parent owns this gate, including after inline fallback. Generated tests stay uncommitted until Step 15. Use Step 7's remaining generation allowance; supply it and the remaining gaps to the same audit prompt. At the cap, omit A and recommend stopping; the listed risk choices remain available.
|
||||
The parent owns this gate, including after inline fallback. Generated tests stay uncommitted until Step 15. The gate only asks; it never hard-fails. Use Step 7's remaining generation allowance; supply it and the remaining gaps to the same audit prompt. At the cap, omit A's generation pass and recommend stopping; A then only lists proposals and the listed risk choices remain available.
|
||||
|
||||
Before proceeding, check CLAUDE.md for a `## Test Coverage` section with `Minimum:` and `Target:` fields. If found, use those percentages. Otherwise use defaults: Minimum = 60%, Target = 80%.
|
||||
Read CLAUDE.md's `## Test Coverage` section for `Minimum:` and `Target:`; otherwise use defaults: Minimum = 60%, Target = 80%. Also read the optional `Generation cap:` (tests per pass, default 5), `Base control:` (`auto` default, or `off`), `Base control budget:` (seconds per run, default 90) and `Star rating:` (`auto` default, or `off`). Missing keys use the defaults.
|
||||
|
||||
Using the coverage percentage from the diagram in substep 4 (the `COVERAGE: X/Y (Z%)` line):
|
||||
**Gate number X.** Take the first matching row; never substitute 0:
|
||||
|
||||
- **>= target:** Pass. "Coverage gate: PASS ({X}%)." Continue.
|
||||
| Step 7 result | Gate number | Print |
|
||||
|---|---|---|
|
||||
| Rating dispatch failed or timed out | skip the gate | rating unavailable: the read-only rating dispatch failed or timed out, so the coverage gate is skipped for this run. Re-run Step 7 to re-rate the tests. (see $GSTACK_ROOT/docs/test-value-bar.md#rating-unavailable) |
|
||||
| Zero paths, test-only diff, or `coverage_pct` null or unparseable | skip the gate | "Coverage gate: could not determine percentage — skipping." |
|
||||
| `Star rating: off` | `coverage_pct` | "Star rating off: gate uses coverage_pct; weak paths still listed." |
|
||||
| `coverage_pct_value` missing, not a number, or outside 0..100 | `coverage_pct` | value-weighted coverage unavailable (outdated installed skill); run /gstack-upgrade. The gate used coverage_pct (any test) this run. (see $GSTACK_ROOT/docs/test-value-bar.md#value-weighted-coverage-unavailable) |
|
||||
| `coverage_pct_value` > `coverage_pct` | `coverage_pct` (clamped) | inconsistent coverage inputs: coverage_pct_value was above coverage_pct, so it was clamped to coverage_pct. Re-run Step 7 if the numbers look wrong. (see $GSTACK_ROOT/docs/test-value-bar.md#inconsistent-coverage-inputs) |
|
||||
| Otherwise | `coverage_pct_value` | — |
|
||||
|
||||
Y is `coverage_pct`; W is `weak_gaps.length`; N is `gaps`. Remaining slots = 2 × generation cap − tests added or extended so far, and 0 once both passes are used. Option A reads "A) Strengthen the existing ★ test for each weak path and generate tests for true gaps ({slots} of {2 × cap} generation slots remaining)"; at 0 slots it reads "A) List the remaining gaps as proposed tests in the PR body" and dispatches nothing.
|
||||
|
||||
- **>= target:** Pass. "Coverage gate: PASS ({X}% value-weighted)." Continue; list weak paths in the PR body as proposed strengthening.
|
||||
- **>= minimum, < target:** Use AskUserQuestion:
|
||||
- "AI-assessed coverage is {X}%. {N} code paths are untested. Target is {target}%."
|
||||
- RECOMMENDATION: Choose A because untested code paths are where production bugs hide.
|
||||
- "Value-weighted coverage is {X}% ({Y}% including {W} weakly covered paths). {W} paths have only weak tests and {N} have none. Target is {target}%."
|
||||
- RECOMMENDATION: Choose A because weakly covered and untested paths are where regressions slip through.
|
||||
- Options:
|
||||
A) Generate more tests for remaining gaps (recommended)
|
||||
A) (as above, recommended)
|
||||
B) Ship anyway — I accept the coverage risk
|
||||
C) These paths don't need tests — mark as intentionally uncovered
|
||||
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B/C or stop; never another generation pass.
|
||||
C) These paths don't need tests — mark as intentionally uncovered. Repo-wide sweep: run /test-audit.
|
||||
- If A and allowance remains: dispatch one generation pass with the weak paths and gaps, then re-evaluate here. At the cap, offer only B/C or stop, plus A as the proposals list; never another generation pass.
|
||||
- If B: Continue. Include in PR body: "Coverage gate: {X}% — user accepted risk."
|
||||
- If C: Continue. Include in PR body: "Coverage gate: {X}% — {N} paths intentionally uncovered."
|
||||
|
||||
- **< minimum:** Use AskUserQuestion:
|
||||
- "AI-assessed coverage is critically low ({X}%). {N} of {M} code paths have no tests. Minimum threshold is {minimum}%."
|
||||
- RECOMMENDATION: Choose A because less than {minimum}% means more code is untested than tested.
|
||||
- "Value-weighted coverage is critically low ({X}%; {Y}% including {W} weakly covered paths). {N} of {M} code paths have no tests. Minimum threshold is {minimum}%."
|
||||
- RECOMMENDATION: Choose A because less than {minimum}% means more behavior is unprotected than protected.
|
||||
- Options:
|
||||
A) Generate tests for remaining gaps (recommended)
|
||||
A) (as above, recommended)
|
||||
B) Override — ship with low coverage (I understand the risk)
|
||||
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B or stop; never another generation pass.
|
||||
- If A and allowance remains: dispatch one generation pass, then re-evaluate here. At the cap, offer only B or stop, plus A as the proposals list; never another generation pass.
|
||||
- If B: Continue. Include in PR body: "Coverage gate: OVERRIDDEN at {X}%."
|
||||
|
||||
**Coverage percentage undetermined:** If the coverage diagram doesn't produce a clear numeric percentage (ambiguous output, parse error), **skip the gate** with: "Coverage gate: could not determine percentage — skipping." Do not default to 0% or block.
|
||||
|
||||
**Test-only diffs:** Skip the gate (same as the existing fast-path).
|
||||
**Spawned or non-interactive session** (the preamble echoed `SESSION_KIND: spawned` or `headless`): ask nothing. Take A restricted to true `gaps` within the remaining slots; never edit tests for weak paths there. List weak paths in the PR body as proposed strengthening.
|
||||
|
||||
**100% coverage:** "Coverage gate: PASS (100%)." Continue.
|
||||
|
||||
@@ -3449,6 +3556,11 @@ theme, excluding VERSION/CHANGELOG bookkeeping. Do not paste the commit list.>
|
||||
## Test Coverage
|
||||
<coverage diagram from Step 7, or "All new code paths have test coverage.">
|
||||
<If Step 7 ran: "Tests: {before} → {after} (+{delta} new)">
|
||||
<If Step 7 ran: "Coverage: {X}% value-weighted ({Y}% including {W} weakly covered paths)">
|
||||
<If Step 7 ran: "Test value: {K} tests written, {R} rejected by the authoring gate, {E} existing tests extended, {W} paths weakly covered (weak = ★, gate-failing or unrated)." Use the singular noun for a count of 1 ("1 test written", "1 existing test extended", "1 path weakly covered").>
|
||||
<Weak paths and leftover gaps as proposed tests with value cards; each regression test's
|
||||
"Regression proof — fails at HEAD · passes at base · passes after fix" line; Test value
|
||||
details for cards whose file type has no known comment syntax.>
|
||||
|
||||
## Pre-Landing Review
|
||||
<findings from Step 9 code review, or "No issues found.">
|
||||
@@ -3585,11 +3697,14 @@ Log metrics for `/retro` through `gstack-review-log`; it handles project/branch
|
||||
JSON validation, storage and sync. It takes **no path argument**; do not build one.
|
||||
|
||||
```bash
|
||||
$GSTACK_ROOT/bin/gstack-review-log '{"skill":"ship","timestamp":"'"$(date -u +%Y-%m-%dT%H:%M:%SZ)"'","coverage_pct":COVERAGE_PCT,"plan_items_total":PLAN_TOTAL,"plan_items_done":PLAN_DONE,"verification_result":"VERIFY_RESULT","version":"VERSION","branch":"'"$(git rev-parse --abbrev-ref HEAD)"'"}'
|
||||
$GSTACK_ROOT/bin/gstack-review-log '{"skill":"ship","timestamp":"'"$(date -u +%Y-%m-%dT%H:%M:%SZ)"'","coverage_pct":COVERAGE_PCT,"coverage_schema":2,"coverage_pct_value":COVERAGE_PCT_VALUE,"weak_gaps":WEAK_GAPS,"tests_extended":TESTS_EXTENDED,"tests_rejected":TESTS_REJECTED,"regression_proof":REGRESSION_PROOF,"plan_items_total":PLAN_TOTAL,"plan_items_done":PLAN_DONE,"verification_result":"VERIFY_RESULT","version":"VERSION","branch":"'"$(git rev-parse --abbrev-ref HEAD)"'"}'
|
||||
```
|
||||
|
||||
Substitute from earlier steps:
|
||||
- **COVERAGE_PCT**: Step 7 diagram's integer percentage; encode null/undetermined as -1
|
||||
- **COVERAGE_PCT_VALUE**: Step 7's `coverage_pct_value` (the gate's X) as an integer, or `null` when missing or ignored
|
||||
- **WEAK_GAPS**, **TESTS_EXTENDED**, **TESTS_REJECTED**: counts of Step 7's `weak_gaps`, `tests_extended` and `tests_rejected` (0 when the key is missing or ignored)
|
||||
- **REGRESSION_PROOF**: `{"red_at_head":N,"base_green":N,"base_unavailable":N}` from Step 7, or `null` when missing
|
||||
- **PLAN_TOTAL**: total plan items extracted in Step 8 (0 if no plan file)
|
||||
- **PLAN_DONE**: count of DONE + CHANGED items from Step 8 (0 if no plan file)
|
||||
- **VERIFY_RESULT**: "pass", "fail", or "skipped", set after Step 9 executes Step 8.1's verification list
|
||||
|
||||
@@ -1133,7 +1133,7 @@ describe('TEST_COVERAGE_AUDIT placeholders', () => {
|
||||
// Regression guard: ship output contains key phrases from before the refactor
|
||||
test('ship SKILL.md regression guard — key phrases preserved', () => {
|
||||
const regressionPhrases = [
|
||||
'100% coverage is the goal',
|
||||
'Coverage goal: every changed behavior is protected by a test that would catch a real regression.',
|
||||
'ASCII coverage diagram',
|
||||
'processPayment',
|
||||
'refundPayment',
|
||||
|
||||
@@ -166,7 +166,7 @@ export const CARVE_GUARDS: Record<string, CarveGuard> = {
|
||||
// wave's headline capability) grows the union to 1.195x. Deliberate:
|
||||
// the section is on-demand (loads only for Apple store targets), so
|
||||
// per-invocation cost for non-iOS ships is one manifest line.
|
||||
maxSizeRatio: 1.322, // Shared advisory identity/dedup + critical-severity validation: 248,065 union bytes / 187,706 baseline = 1.3216 (2026-09-17).
|
||||
maxSizeRatio: 1.397, // Shared advisory identity/dedup + critical-severity validation: 248,065 union bytes / 187,706 baseline = 1.3216 (2026-09-17). + test value bar in the lazy Step 7 section (value cards, weak paths, gate table, base control, machine checks; ~13.6KB): measured 1.396 (2026-09-29).
|
||||
},
|
||||
'plan-ceo-review': {
|
||||
skill: 'plan-ceo-review',
|
||||
@@ -221,7 +221,7 @@ export const CARVE_GUARDS: Record<string, CarveGuard> = {
|
||||
// 1.08 → 1.10: the scope-gate exceptions block (+ its adversarial-review
|
||||
// hardening: host-anchored mode signal, precedence, passing-mention
|
||||
// guards) and the plan-mode preamble reword land the union at 1.092.
|
||||
maxSizeRatio: 1.151, // + clarity rules for saved decisions/setup gates + the Aside probe's failure reason; measured 1.1504
|
||||
maxSizeRatio: 1.169, // + clarity rules for saved decisions/setup gates + the Aside probe's failure reason; measured 1.1504. + test value bar and Tests to Retire in the lazy Test review section (~2.6KB); measured 1.168
|
||||
},
|
||||
'plan-design-review': {
|
||||
skill: 'plan-design-review',
|
||||
@@ -664,7 +664,7 @@ do not launch the downstream skill or open a browser.`,
|
||||
},
|
||||
behavioral: 'prompt',
|
||||
maxSkeletonBytes: 63_500, // + v2.0 {{ASIDE_SETUP}}/{{BROWSE_FALLBACK}} (replaces the browse setup block); measured 61_253
|
||||
maxSizeRatio: 1.08, // + v1.81 Aside contract + gstack-browser fallback block; measured 1.063
|
||||
maxSizeRatio: 1.095, // + v1.81 Aside contract + gstack-browser fallback block (1.080 on v1.91.7.0) + the shared test value bar at 8a.5 ({{TEST_VALUE_BAR:qa}}); measured 1.094
|
||||
minUnionBytes: 69_500, // measured union 70,385
|
||||
// 'aside repl' pins the Aside contract; '$B goto' pins the fallback block in the always-loaded skeleton.
|
||||
mustContain: ['bug', 'aside repl', '$B goto', 'fix', 'Health Score Rubric', 'regression'],
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
/**
|
||||
* Fixture for the test value bar evals (test/skill-e2e-test-value.test.ts).
|
||||
*
|
||||
* A tiny Bun project. The feature branch adds three tests:
|
||||
* - test/pricing-source.test.ts greps src/pricing.ts for a function name
|
||||
* (low value: exact source grep, no declared contract);
|
||||
* - test/pricing-reset.test.ts exists only to call _resetPriceCacheForTests,
|
||||
* a test-only export with no production caller (low value);
|
||||
* - test/skill-golden.test.ts compares generated SKILL.md bytes with a golden
|
||||
* file (a generated-output contract the retention bar keeps).
|
||||
*/
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
|
||||
export const LOW_VALUE_TESTS = {
|
||||
sourceGrep: 'test/pricing-source.test.ts',
|
||||
testOnlyExport: 'test/pricing-reset.test.ts',
|
||||
golden: 'test/skill-golden.test.ts',
|
||||
testOnlySymbol: '_resetPriceCacheForTests',
|
||||
} as const;
|
||||
|
||||
function write(dir: string, relative: string, content: string): void {
|
||||
fs.mkdirSync(path.dirname(path.join(dir, relative)), { recursive: true });
|
||||
fs.writeFileSync(path.join(dir, relative), content);
|
||||
}
|
||||
|
||||
function git(dir: string, args: string[]): void {
|
||||
const result = spawnSync('git', args, { cwd: dir, stdio: 'pipe', timeout: 10_000 });
|
||||
if (result.status !== 0) throw new Error(`git ${args.join(' ')} failed: ${result.stderr}`);
|
||||
}
|
||||
|
||||
export function createTestValueFixture(dir: string): void {
|
||||
write(dir, 'package.json', JSON.stringify({ name: 'pricing-app', version: '1.0.0', type: 'module', scripts: { test: 'bun test' } }, null, 2) + '\n');
|
||||
write(dir, 'src/pricing.ts', `const cache = new Map<string, number>();
|
||||
|
||||
export function applyDiscount(price: number, percent: number): number {
|
||||
if (percent < 0 || percent > 100) throw new Error('Invalid discount');
|
||||
const key = \`\${price}:\${percent}\`;
|
||||
if (!cache.has(key)) cache.set(key, Math.round(price * (100 - percent)) / 100);
|
||||
return cache.get(key)!;
|
||||
}
|
||||
`);
|
||||
write(dir, 'src/checkout.ts', `import { applyDiscount } from './pricing';
|
||||
|
||||
export function total(prices: number[], percent: number): number {
|
||||
return prices.reduce((sum, price) => sum + applyDiscount(price, percent), 0);
|
||||
}
|
||||
`);
|
||||
write(dir, 'scripts/gen-skill.ts', `export function renderSkill(name: string): string {
|
||||
return \`# \${name}\\n\\nRun \\\`/\${name}\\\` to start.\\n\`;
|
||||
}
|
||||
`);
|
||||
write(dir, 'test/fixtures/golden/SKILL.md', '# pricing\n\nRun `/pricing` to start.\n');
|
||||
write(dir, 'test/pricing.test.ts', `import { test, expect } from 'bun:test';
|
||||
import { applyDiscount } from '../src/pricing';
|
||||
import { total } from '../src/checkout';
|
||||
|
||||
test('applies a discount and rejects an invalid percent', () => {
|
||||
expect(applyDiscount(100, 25)).toBe(75);
|
||||
expect(() => applyDiscount(100, 101)).toThrow('Invalid discount');
|
||||
expect(total([100, 50], 10)).toBe(135);
|
||||
});
|
||||
`);
|
||||
write(dir, 'CLAUDE.md', '# pricing-app\n\n## Testing\n\nRun `bun test`.\n');
|
||||
git(dir, ['init', '-q', '-b', 'main']);
|
||||
git(dir, ['config', 'user.email', 'test@test.com']);
|
||||
git(dir, ['config', 'user.name', 'Test']);
|
||||
git(dir, ['add', '.']);
|
||||
git(dir, ['commit', '-q', '-m', 'initial pricing app']);
|
||||
git(dir, ['checkout', '-q', '-b', 'feature/pricing-cache']);
|
||||
|
||||
fs.appendFileSync(path.join(dir, 'src/pricing.ts'), `
|
||||
export function ${LOW_VALUE_TESTS.testOnlySymbol}(): void {
|
||||
cache.clear();
|
||||
}
|
||||
`);
|
||||
write(dir, LOW_VALUE_TESTS.sourceGrep, `import { test, expect } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
|
||||
test('pricing module defines applyDiscount', () => {
|
||||
const source = fs.readFileSync(new URL('../src/pricing.ts', import.meta.url), 'utf8');
|
||||
expect(source).toContain('export function applyDiscount');
|
||||
});
|
||||
`);
|
||||
write(dir, LOW_VALUE_TESTS.testOnlyExport, `import { test, expect } from 'bun:test';
|
||||
import { ${LOW_VALUE_TESTS.testOnlySymbol} } from '../src/pricing';
|
||||
|
||||
test('cache reset helper is callable', () => {
|
||||
expect(() => ${LOW_VALUE_TESTS.testOnlySymbol}()).not.toThrow();
|
||||
});
|
||||
`);
|
||||
write(dir, LOW_VALUE_TESTS.golden, `import { test, expect } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import { renderSkill } from '../scripts/gen-skill';
|
||||
|
||||
test('generated SKILL.md matches the golden bytes', () => {
|
||||
const golden = fs.readFileSync(new URL('./fixtures/golden/SKILL.md', import.meta.url), 'utf8');
|
||||
expect(renderSkill('pricing')).toBe(golden);
|
||||
});
|
||||
`);
|
||||
git(dir, ['add', '.']);
|
||||
git(dir, ['commit', '-q', '-m', 'add pricing cache and tests']);
|
||||
}
|
||||
|
||||
export function lastJsonLine(output: string): any {
|
||||
const lines = output.trim().split('\n').map(line => line.trim()).filter(Boolean);
|
||||
for (let index = lines.length - 1; index >= 0; index--) {
|
||||
const line = lines[index]!.replace(/^`+|`+$/g, '');
|
||||
if (!line.startsWith('{')) continue;
|
||||
try { return JSON.parse(line); } catch {}
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export function jsonFindings(output: string): any[] {
|
||||
return output.split('\n').map(line => line.trim()).filter(line => line.startsWith('{') && line.includes('"severity"')).flatMap(line => {
|
||||
try { return [JSON.parse(line)]; } catch { return []; }
|
||||
});
|
||||
}
|
||||
@@ -783,6 +783,10 @@ export const E2E_TOUCHFILES: Record<string, string[]> = {
|
||||
"test/fixtures/coverage-diagram-legend-as.json",
|
||||
'test/helpers/coverage-audit.ts', 'test/helpers/office-hours-attempt.ts'
|
||||
],
|
||||
// Test value bar behavior (weak paths, low-value findings, report-only sweep)
|
||||
'ship-coverage-value': ['ship/**', 'scripts/resolvers/testing.ts', 'test/fixtures/coverage-audit-fixture.ts', 'test/skill-e2e-test-value.test.ts', 'test/helpers/test-value-fixture.ts', 'scripts/resolvers/test-value.ts', 'test/helpers/office-hours-attempt.ts', 'lib/eval-model.ts'],
|
||||
'review-test-value': ['review/**', 'test/fixtures/coverage-audit-fixture.ts', 'test/skill-e2e-test-value.test.ts', 'test/helpers/test-value-fixture.ts', 'scripts/resolvers/test-value.ts', 'test/helpers/office-hours-attempt.ts', 'lib/eval-model.ts'],
|
||||
'test-audit-report-only': ['test-audit/**', 'test/fixtures/coverage-audit-fixture.ts', 'test/skill-e2e-test-value.test.ts', 'test/helpers/test-value-fixture.ts', 'scripts/resolvers/test-value.ts', 'test/helpers/office-hours-attempt.ts', 'lib/eval-model.ts'],
|
||||
'plan-eng-coverage-audit': [
|
||||
'scripts/resolvers/learnings.ts',
|
||||
'test/fixtures/coverage-audit-ci-diagrams.json',
|
||||
@@ -1231,6 +1235,9 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic'> = {
|
||||
'plan-eng-review': 'periodic',
|
||||
'plan-eng-review-artifact': 'periodic',
|
||||
'plan-eng-coverage-audit': 'gate',
|
||||
'ship-coverage-value': 'gate',
|
||||
'review-test-value': 'gate',
|
||||
'test-audit-report-only': 'gate',
|
||||
'plan-review-report': 'gate',
|
||||
|
||||
// Plan-mode handshake. plan-ceo/plan-devex ask-first reliably (gate-tier);
|
||||
|
||||
@@ -184,7 +184,7 @@ test('detached PR fallback and release commands cover their actual default worke
|
||||
const prFloor = Math.ceil((Math.ceil(fullGateFiles.length / prWorkers) * 1_800_000 + fullGateFiles.reduce(
|
||||
(total, file) => total + Math.max(0, resolvePaidShardBudget([file]).timeoutMs - 1_800_000), 0,
|
||||
)) / 1000 * 1.05);
|
||||
expect(prFloor).toBe(74_981);
|
||||
expect(prFloor).toBe(76_871);
|
||||
expect(prWall).toBe(92_820_000);
|
||||
expect(prWall).toBeGreaterThanOrEqual(paidShardWallUpperBoundMs(files, prWorkers) + 120_000);
|
||||
|
||||
@@ -236,9 +236,9 @@ test('both gate executors cover the complete census without increasing aggregate
|
||||
expect(executor.strategy.matrix.slice).toEqual(Array.from({ length: slices }, (_, i) => i + 1));
|
||||
expect(planned.slices).toBe(slices);
|
||||
const manifest = buildRunManifest({ tier: 'gate', sliceCount: planned.slices, evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(46);
|
||||
expect(manifest.entries.filter(row => row.status === 'planned')).toHaveLength(47);
|
||||
const files = manifest.entries.filter(row => row.status === 'planned').map(row => row.file);
|
||||
expect(new Set(files).size).toBe(46);
|
||||
expect(new Set(files).size).toBe(47);
|
||||
expect(files).toContain('test/skill-e2e-ship-skip.test.ts');
|
||||
expect(files.sort()).toEqual(selectPaidTestFiles(collectPaidTestFiles(), 'gate').selected.sort());
|
||||
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
|
||||
@@ -246,7 +246,7 @@ test('both gate executors cover the complete census without increasing aggregate
|
||||
));
|
||||
expect(executor['timeout-minutes'] * 60_000).toBeGreaterThanOrEqual(Math.max(...walls) + 20 * 60_000);
|
||||
if (jobName === 'gate-census') {
|
||||
expect(Math.max(...walls)).toBe(16_440_000);
|
||||
expect(Math.max(...walls)).toBe(17_020_000);
|
||||
expect(executor['timeout-minutes']).toBe(352);
|
||||
expect(emit[0].env.EVALS_ALL).toBe('1');
|
||||
expect(executor.strategy['max-parallel']).toBe(4);
|
||||
@@ -280,7 +280,7 @@ test('the periodic executor supervises every actual case and retry within its CI
|
||||
const manifest = buildRunManifest({ tier: 'periodic', sliceCount: planned.slices,
|
||||
evalsAll: true, env: { EVALS_ALL: '1' } });
|
||||
const census = manifest.entries.filter(row => row.status === 'planned');
|
||||
expect(census).toHaveLength(70);
|
||||
expect(census).toHaveLength(71);
|
||||
expect(census.find(row => row.file === 'test/skill-llm-eval.test.ts')?.budget?.timeoutMs).toBe(6_220_000);
|
||||
const walls = executor.strategy.matrix.slice.map((slice: number) => paidShardWallUpperBoundMs(
|
||||
census.filter(row => row.slice === slice).map(row => row.file), active.jobs,
|
||||
|
||||
@@ -13,7 +13,10 @@ const periodicIds = Object.keys(E2E_TOUCHFILES).filter(id => E2E_TIERS[id] === '
|
||||
const judgeIds = Object.keys(LLM_JUDGE_TOUCHFILES).sort();
|
||||
|
||||
test.each(sharedInputs)('%s retains the full gate after native dependency registration', file => {
|
||||
const result = computePaidCaseSelection({ profile: 'pr', env: {}, changedFiles: [file] });
|
||||
// An unresolvable base keeps this independent of the checkout: a local
|
||||
// package.json identical to its merge-base would otherwise count as a
|
||||
// version-only change and be dropped before the shared-input rule.
|
||||
const result = computePaidCaseSelection({ profile: 'pr', env: { EVALS_BASE: 'refs/heads/no-such-base' }, changedFiles: [file] });
|
||||
expect(result.coverage?.mode).toBe('full-fallback');
|
||||
expect(result.selection.e2e).toEqual(gateIds);
|
||||
expect(result.selection.judges).toEqual(judgeIds);
|
||||
|
||||
@@ -351,7 +351,7 @@ describe('QA caller authority in pure host renders', () => {
|
||||
expect(body.match(/If A and allowance remains:/g)).toHaveLength(2);
|
||||
expect(body).toContain('At the cap, offer only B/C or stop');
|
||||
expect(body).toContain('At the cap, offer only B or stop');
|
||||
expect(body).toContain('At the cap, omit A and recommend stopping');
|
||||
expect(body).toContain('At the cap, omit A\'s generation pass and recommend stopping');
|
||||
expect(body).toContain('Minimum = 60%, Target = 80%');
|
||||
expect(body).not.toContain('Maximum 2 passes total');
|
||||
});
|
||||
|
||||
@@ -275,7 +275,7 @@ test('ship template consolidation: initial, failed and inline generation attempt
|
||||
expect(allowance).toContain('Re-entry never resets it');
|
||||
expect(allowance).toContain('Two passes already used means no further generation');
|
||||
expect(allowance).toContain('read-only reassessment uses no pass');
|
||||
expect(allowance).toContain('30-path/20-test/2-minute per-test caps');
|
||||
expect(allowance).toContain('30-path/5-tests-per-pass/2-minute per-test caps');
|
||||
expect(allowance).toContain('missing permission is not approval');
|
||||
expect(coverageTemplate).toContain("confirm it stopped before running the same audit inline");
|
||||
});
|
||||
|
||||
@@ -0,0 +1,156 @@
|
||||
/** Test value bar behavior in /ship Step 7, the /review testing specialist and
|
||||
* /test-audit. Each case is one bounded capture on a small fixture:
|
||||
* ship-coverage-value a ★-only path lands in weak_gaps and X <= Y
|
||||
* review-test-value a source-grep test and a test-only export are
|
||||
* INFORMATIONAL findings; the SKILL.md golden is not
|
||||
* test-audit-report-only the same fixture yields complete retirement cards,
|
||||
* keeps the golden and edits nothing
|
||||
*/
|
||||
import { afterAll } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { CAPTURE_MS } from './helpers/eval-budgets';
|
||||
import { runSkillTest } from './helpers/session-runner';
|
||||
import { ROOT, runId, describeIfSelected, testIfSelected, copyDirSync, logCost,
|
||||
createEvalCollector, finalizeEvalCollector } from './helpers/e2e-helpers';
|
||||
import { extractSkillBody } from './helpers/skill-fixture';
|
||||
import { createCoverageAuditFixture } from './fixtures/coverage-audit-fixture';
|
||||
import { createTestValueFixture, lastJsonLine, jsonFindings, LOW_VALUE_TESTS } from './helpers/test-value-fixture';
|
||||
import { runRecordedOfficeHoursAttempt, OFFICE_HOURS_BUN_GRACE_MS } from './helpers/office-hours-attempt';
|
||||
import { resolveEvalModel } from '../lib/eval-model';
|
||||
import { RETIREMENT_FIELDS } from '../scripts/resolvers/test-value';
|
||||
import type { SkillTestResult } from './helpers/session-runner';
|
||||
|
||||
const evalCollector = createEvalCollector('e2e');
|
||||
const RUNNER_MS = CAPTURE_MS - 3 * OFFICE_HOURS_BUN_GRACE_MS;
|
||||
|
||||
function trackedChanges(cwd: string): string {
|
||||
return spawnSync('git', ['status', '--porcelain', '--untracked-files=no'], { cwd, encoding: 'utf8', timeout: 10_000 }).stdout.trim();
|
||||
}
|
||||
|
||||
const SMOKE_TEST = `
|
||||
import { refundPayment } from '../src/billing';
|
||||
test('refundPayment does not throw', () => {
|
||||
expect(() => refundPayment('pay_1', 'duplicate')).not.toThrow();
|
||||
});
|
||||
`;
|
||||
|
||||
type Case = {
|
||||
id: string;
|
||||
skill: string;
|
||||
suite: string;
|
||||
setup: (cwd: string) => void;
|
||||
prompt: (cwd: string, reportDir: string) => string;
|
||||
validate: (result: SkillTestResult, cwd: string, reportDir: string) => void;
|
||||
};
|
||||
|
||||
const CASES: Case[] = [
|
||||
{
|
||||
id: 'ship-coverage-value', skill: 'ship', suite: 'Ship Test Value E2E',
|
||||
setup: cwd => {
|
||||
createCoverageAuditFixture(cwd);
|
||||
fs.appendFileSync(path.join(cwd, 'test/billing.test.ts'), SMOKE_TEST);
|
||||
},
|
||||
prompt: cwd => `Read ship/SKILL.md and ship/sections/test-coverage.md for the current ship workflow.
|
||||
|
||||
You are on the feature/billing branch. The base branch is main. There is no remote.
|
||||
Run ONLY the Step 7 coverage audit, inline, as the audit subagent would, applying it
|
||||
directly to ${cwd}/src/billing.ts and its tests in ${cwd}/test/billing.test.ts (a
|
||||
targeted audit with no branch diff). Generation: audit-only; passes used: 0 of 2.
|
||||
Do not dispatch subagents, write tests or modify files. Skip every other step.
|
||||
End with the audit's LAST-line JSON exactly as Step 7 specifies.`,
|
||||
validate: result => {
|
||||
const json = lastJsonLine(result.output || '');
|
||||
if (!json) throw new Error('ship coverage: no last-line JSON');
|
||||
if (typeof json.coverage_pct !== 'number' || typeof json.coverage_pct_value !== 'number') throw new Error(`ship coverage: numeric coverage_pct and coverage_pct_value required, got ${JSON.stringify(json)}`);
|
||||
if (json.coverage_pct < json.coverage_pct_value) throw new Error(`ship coverage: coverage_pct ${json.coverage_pct} < coverage_pct_value ${json.coverage_pct_value}`);
|
||||
const weak = Array.isArray(json.weak_gaps) ? json.weak_gaps : [];
|
||||
if (!weak.some((gap: any) => /refund/i.test(JSON.stringify(gap)))) throw new Error(`ship coverage: the ★-only refundPayment path must be in weak_gaps, got ${JSON.stringify(json.weak_gaps)}`);
|
||||
},
|
||||
},
|
||||
{
|
||||
id: 'review-test-value', skill: 'review', suite: 'Review Test Value E2E',
|
||||
setup: cwd => createTestValueFixture(cwd),
|
||||
prompt: () => `Read review/SKILL.md and review/sections/review-army.md for the current review workflow.
|
||||
Apply ONLY the testing specialist checklist in review/specialists/testing.md to this
|
||||
branch's diff (\`git diff main...HEAD\`), as the testing specialist would. Do not
|
||||
dispatch other specialists, fix anything or modify files. Output the specialist's JSON
|
||||
findings, one per line.`,
|
||||
validate: (result, cwd) => {
|
||||
const findings = jsonFindings(result.output || '');
|
||||
const about = (needle: RegExp) => findings.filter(finding => needle.test(`${finding.path} ${finding.summary}`));
|
||||
const grep = about(/pricing-source/);
|
||||
const seam = about(new RegExp(`pricing-reset|${LOW_VALUE_TESTS.testOnlySymbol}`));
|
||||
if (!grep.length || !grep.every(finding => finding.severity === 'INFORMATIONAL')) throw new Error(`review: source-grep test needs an INFORMATIONAL finding, got ${JSON.stringify(grep)}`);
|
||||
if (!seam.length || !seam.every(finding => finding.severity === 'INFORMATIONAL')) throw new Error(`review: test-only export needs an INFORMATIONAL finding, got ${JSON.stringify(seam)}`);
|
||||
if (!seam.some(finding => /non_test_callers/.test(JSON.stringify(finding.evidence ?? '')) && /git grep/.test(JSON.stringify(finding.evidence ?? '')))) throw new Error('review: test-only export evidence must record non_test_callers and the git grep search');
|
||||
const golden = about(/skill-golden/);
|
||||
if (golden.length) throw new Error(`review: the SKILL.md golden test must not be flagged, got ${JSON.stringify(golden)}`);
|
||||
if (trackedChanges(cwd)) throw new Error('review: tracked files changed');
|
||||
},
|
||||
},
|
||||
{
|
||||
id: 'test-audit-report-only', skill: 'test-audit', suite: 'Test Audit Report-Only E2E',
|
||||
setup: cwd => createTestValueFixture(cwd),
|
||||
prompt: (_cwd, reportDir) => `Read test-audit/SKILL.md and run /test-audit on this repository, report-only.
|
||||
Treat this as a headless session: ask no questions, approve no batch, edit no file in
|
||||
the repository. There is no gstack install here, so skip the SLUG setup line and write
|
||||
the report to ${reportDir}/test-audit.md and its JSON sidecar to
|
||||
${reportDir}/test-audit.json instead. Stop after Step 4.`,
|
||||
validate: (_result, cwd, reportDir) => {
|
||||
const sidecarPath = path.join(reportDir, 'test-audit.json');
|
||||
if (!fs.existsSync(path.join(reportDir, 'test-audit.md'))) throw new Error('test-audit: report missing');
|
||||
if (!fs.existsSync(sidecarPath)) throw new Error('test-audit: JSON sidecar missing');
|
||||
const sidecar = JSON.parse(fs.readFileSync(sidecarPath, 'utf8'));
|
||||
const candidates: any[] = Array.isArray(sidecar.candidates) ? sidecar.candidates : [];
|
||||
const retiring = candidates.filter(candidate => candidate.verdict !== 'retain');
|
||||
for (const needle of [/pricing-source/, new RegExp(`pricing-reset|${LOW_VALUE_TESTS.testOnlySymbol}`)]) {
|
||||
const match = retiring.find(candidate => needle.test(String(candidate.test)));
|
||||
if (!match) throw new Error(`test-audit: missing candidate ${needle}, got ${JSON.stringify(candidates.map(c => c.test))}`);
|
||||
const missing = RETIREMENT_FIELDS.filter(field => !String(match.retirement_card?.[field] ?? '').trim());
|
||||
if (missing.length) throw new Error(`test-audit: ${match.test} retirement card missing ${missing.join(', ')}`);
|
||||
}
|
||||
if (retiring.some(candidate => /skill-golden/.test(String(candidate.test)))) throw new Error('test-audit: the SKILL.md golden must be retained');
|
||||
if (trackedChanges(cwd)) throw new Error('test-audit: tracked files changed');
|
||||
},
|
||||
},
|
||||
];
|
||||
|
||||
for (const entry of CASES) describeIfSelected(entry.suite, [entry.id], () => {
|
||||
testIfSelected(entry.id, async () => {
|
||||
let cwd: string | undefined;
|
||||
let reportDir: string | undefined;
|
||||
try {
|
||||
await runRecordedOfficeHoursAttempt({
|
||||
collector: evalCollector, name: entry.id, suite: entry.suite,
|
||||
model: process.env.EVALS_MODEL ?? resolveEvalModel('capture'),
|
||||
budgetMs: CAPTURE_MS - OFFICE_HOURS_BUN_GRACE_MS,
|
||||
run: async signal => {
|
||||
cwd = fs.mkdtempSync(path.join(os.tmpdir(), `skill-e2e-${entry.id}-`));
|
||||
reportDir = fs.mkdtempSync(path.join(os.tmpdir(), `skill-e2e-${entry.id}-report-`));
|
||||
entry.setup(cwd);
|
||||
copyDirSync(path.join(ROOT, entry.skill), path.join(cwd, entry.skill));
|
||||
fs.writeFileSync(path.join(cwd, entry.skill, 'SKILL.md'), extractSkillBody(path.join(ROOT, entry.skill)));
|
||||
fs.writeFileSync(path.join(cwd, '.git', 'info', 'exclude'), `${entry.skill}/\n`);
|
||||
return runSkillTest({
|
||||
prompt: entry.prompt(cwd, reportDir),
|
||||
workingDirectory: cwd, maxTurns: 25,
|
||||
allowedTools: ['Bash', 'Read', 'Write', 'Glob', 'Grep'],
|
||||
timeout: RUNNER_MS, testName: entry.id, runId, signal,
|
||||
});
|
||||
},
|
||||
validate: result => {
|
||||
logCost(entry.id, result);
|
||||
if (result.exitReason !== 'success') throw new Error(`${entry.id}: ${result.exitReason}`);
|
||||
entry.validate(result, cwd!, reportDir!);
|
||||
},
|
||||
});
|
||||
} finally {
|
||||
for (const dir of [cwd, reportDir]) if (dir) try { fs.rmSync(dir, { recursive: true, force: true }); } catch {}
|
||||
}
|
||||
}, CAPTURE_MS);
|
||||
});
|
||||
|
||||
afterAll(async () => { await finalizeEvalCollector(evalCollector); });
|
||||
@@ -250,10 +250,11 @@ describe('SKILL.md size budget regression (gate, free)', () => {
|
||||
// estimate was a moving target: 4177 solo, 8356 and 8041 in two parallel
|
||||
// runs. A repo-budget ratchet measures the catalog that ships; CI always
|
||||
// checks the PR's committed tree anyway.
|
||||
const trackedPaths = execSync('git ls-files -- "*/SKILL.md"', { cwd: REPO_ROOT, encoding: 'utf-8', timeout: 30_000 })
|
||||
// List paths from HEAD too: a staged-but-uncommitted skill is in the index
|
||||
// but not yet in HEAD, so `git ls-files` + `git show HEAD:` disagree.
|
||||
const trackedPaths = execSync('git ls-tree -r --name-only HEAD', { cwd: REPO_ROOT, encoding: 'utf-8', timeout: 30_000, maxBuffer: 16 * 1024 * 1024 })
|
||||
.split('\n')
|
||||
.filter(Boolean)
|
||||
.filter((p) => p.split('/').length === 2);
|
||||
.filter((p) => p.endsWith('/SKILL.md') && p.split('/').length === 2);
|
||||
let descriptionBytes = 0;
|
||||
for (const rel of trackedPaths) {
|
||||
const committed = execSync(`git show HEAD:${JSON.stringify(rel)}`, {
|
||||
|
||||
@@ -1321,9 +1321,9 @@ describe('Step 3.4 test coverage audit', () => {
|
||||
expect(content).toContain('Never commit failing tests');
|
||||
});
|
||||
|
||||
test('Step 3.4 includes vibe coding philosophy', () => {
|
||||
test('Step 3.4 states the value-based coverage goal', () => {
|
||||
const content = readShipUnion();
|
||||
expect(content).toContain('vibe coding becomes yolo coding');
|
||||
expect(content).toContain('Coverage goal: every changed behavior is protected by a test that would catch a real regression. Test count is not a goal.');
|
||||
});
|
||||
|
||||
test('Step 3.4 traces actual codepaths, not just syntax', () => {
|
||||
|
||||
@@ -0,0 +1,241 @@
|
||||
/**
|
||||
* The test value bar ships in every prompt that proposes, writes, reviews or
|
||||
* sweeps tests. These checks prove the prompts carry the contract; the paid
|
||||
* cases in test/skill-e2e-test-value.test.ts carry the behavior claim.
|
||||
*/
|
||||
import { describe, expect, test } from 'bun:test';
|
||||
import * as fs from 'node:fs';
|
||||
import * as os from 'node:os';
|
||||
import * as path from 'node:path';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { HOST_PATHS, type TemplateContext } from '../scripts/resolvers/types';
|
||||
import { generateTestCoverageAuditInner } from '../scripts/resolvers/testing';
|
||||
import {
|
||||
CALLER_SEARCH_COMMAND, CALLER_SYMBOL_PATTERN, CARD_FIELD_MAX_BYTES, CATALOG, MESSAGES, PRAGMA, QUESTIONS, REASON_CODES,
|
||||
RETENTION_ONE_LINER, RETIREMENT_FIELDS, REVIEW_EVIDENCE_FIELDS, SWEEP_POINTER, TEST_VALUE_BAR_MAX_BYTES, TEST_VALUE_BAR_MODES,
|
||||
WEAK_REASONS, clampCardField, generateTestValueBar, renderValueCard, type TestValueBarMode,
|
||||
} from '../scripts/resolvers/test-value';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const FIX = 'Fix: bun run gen:skill-docs && bun test test/test-value-bar.test.ts';
|
||||
const read = (relative: string) => fs.readFileSync(path.join(ROOT, relative), 'utf8');
|
||||
const ctx = { skillName: 'ship', tmplPath: '', host: 'claude', paths: HOST_PATHS.claude } as TemplateContext;
|
||||
const bar = (mode: TestValueBarMode) => generateTestValueBar(ctx, [mode]);
|
||||
|
||||
const CARD_FORMAT = 'Value: protects=<...>; fails_when=<...>; why_new=<...>; seam=none';
|
||||
const shipSection = read('ship/sections/test-coverage.md');
|
||||
const shipGate = generateTestCoverageAuditInner(ctx, 'ship', 'gate');
|
||||
|
||||
const CONTRACT: Record<TestValueBarMode, { render: string; required: string[]; absent?: string[] }> = {
|
||||
plan: {
|
||||
render: generateTestCoverageAuditInner(ctx, 'plan'),
|
||||
required: [
|
||||
...QUESTIONS, CARD_FORMAT, `at most ${CARD_FIELD_MAX_BYTES} UTF-8 bytes`, 'A missing upstream card never blocks',
|
||||
'Coverage goal: every changed behavior is protected by a test that would catch a real regression. Test count is not a goal.',
|
||||
'X = paths with a ★★/★★★ test / total paths', 'Y = paths with any test / total paths', RETENTION_ONE_LINER,
|
||||
'unless exact output is the declared contract (goldens, prompt bytes, wire formats)',
|
||||
'Its value card', 'Tests made obsolete by this plan', '## Tests to Retire',
|
||||
],
|
||||
absent: ['100% coverage is the goal', 'Step 4.75'],
|
||||
},
|
||||
ship: {
|
||||
render: shipSection,
|
||||
required: [
|
||||
...QUESTIONS, CARD_FORMAT, 'Coverage goal: every changed behavior is protected by a test that would catch a real regression.',
|
||||
`\`weak_gaps\` (reason \`${WEAK_REASONS.join('|')}\`), not \`gaps\``, 'zero skips the gate', RETENTION_ONE_LINER,
|
||||
'5 tests per generation pass', 'an extension uses one slot and a rejection uses none', '30-path/5-tests-per-pass/2-minute per-test caps',
|
||||
'precedence extended > added > rejected', REASON_CODES.join(', '),
|
||||
'"coverage_pct":N,"gaps":N,"diagram":"<full markdown coverage diagram for PR body>","tests_added":["path",...],"coverage_pct_value":N,"weak_gaps":[',
|
||||
'a value-card header with four non-empty fields (else\n `incomplete_card`)', 'a later duplicate is `duplicate_protects`', 'is `needs_seam`',
|
||||
'No `tests_rejected` path may remain on disk as a new file', 'dispatch one read-only Agent', 'it uses no generation\n pass',
|
||||
'Regression proof — fails at HEAD: yes · passes at base: yes | unavailable (<reason>) | manual · passes after fix: yes | pending',
|
||||
'base control unavailable: collection error', 'N of M regression tests got base control', 'worktree add --quiet --detach',
|
||||
'trap cleanup EXIT INT TERM', 'timeout 30 git -C "$ROOT" fetch', 'within a 3-minute total per /ship run',
|
||||
"find \"${TMPDIR:-/tmp}\" -maxdepth 1 -name 'gstack-base-control.*'",
|
||||
'K = `tests_added.length`', 'R = `tests_rejected.length`', 'E = `tests_extended.length`', 'W = `weak_gaps.length`',
|
||||
'The gate only asks; it never hard-fails.', `C) These paths don't need tests — mark as intentionally uncovered. ${SWEEP_POINTER}`,
|
||||
'generation slots remaining)', 'A) List the remaining gaps as proposed tests in the PR body',
|
||||
'Take A restricted to true `gaps` within the remaining slots; never edit tests for weak paths there.',
|
||||
],
|
||||
absent: ['100% coverage is the goal', '20 tests generated max', '30-path/20-test'],
|
||||
},
|
||||
qa: {
|
||||
render: bar('qa'),
|
||||
required: [QUESTIONS[2], QUESTIONS[3], CARD_FORMAT, 'Put it in the 8e.5 record (/qa) or under each proposed test (/qa-only).', 'A missing upstream card never blocks'],
|
||||
absent: [QUESTIONS[0], QUESTIONS[1]],
|
||||
},
|
||||
audit: {
|
||||
render: bar('audit'),
|
||||
required: [...QUESTIONS, ...CATALOG, RETIREMENT_FIELDS.map(field => `\`${field}\``).join(', '), CALLER_SEARCH_COMMAND, CALLER_SYMBOL_PATTERN,
|
||||
'caller check unavailable: unsupported symbol', 'grep-only', 'typecheck/build or dead-code tool', PRAGMA, 'Never retire anything reachable from the package entrypoint'],
|
||||
},
|
||||
};
|
||||
|
||||
describe('test value bar render contract', () => {
|
||||
for (const mode of TEST_VALUE_BAR_MODES) test(`${mode}: required contract strings and byte ceiling`, () => {
|
||||
const { render, required, absent = [] } = CONTRACT[mode];
|
||||
const missing = required.filter(text => !render.includes(text));
|
||||
expect(missing, `${mode} render lacks contract text. ${FIX}`).toEqual([]);
|
||||
expect(absent.filter(text => render.includes(text)), `${mode} render keeps retired text. ${FIX}`).toEqual([]);
|
||||
const rendered = bar(mode);
|
||||
expect(rendered.includes('Example: Value: protects=') && rendered.includes('Rejected (covered_elsewhere)'), `${mode} needs one good card and one rejected proposal`).toBe(true);
|
||||
expect(Buffer.byteLength(rendered, 'utf8')).toBeLessThanOrEqual(TEST_VALUE_BAR_MAX_BYTES[mode]);
|
||||
});
|
||||
|
||||
test('plan and ship coverage audits render the bar at their call site', () => {
|
||||
for (const mode of ['plan', 'ship'] as const) expect(generateTestCoverageAuditInner(ctx, mode)).toContain(bar(mode));
|
||||
});
|
||||
|
||||
test('an unknown mode and an over-budget render fail generation with actionable text', () => {
|
||||
expect(() => generateTestValueBar(ctx, ['foo'])).toThrow('Unknown TEST_VALUE_BAR mode foo; expected plan|ship|qa|audit');
|
||||
expect(() => generateTestValueBar(ctx, ['foo'])).toThrow('scripts/resolvers/test-value.ts');
|
||||
const saved = TEST_VALUE_BAR_MAX_BYTES.qa;
|
||||
try {
|
||||
TEST_VALUE_BAR_MAX_BYTES.qa = 10;
|
||||
expect(() => bar('qa')).toThrow(/^TEST_VALUE_BAR mode 'qa' renders \d+ bytes, budget 10 \(over by \d+\)\. Trim the mode's section in scripts\/resolvers\/test-value\.ts or raise the ceiling with a reason\.$/);
|
||||
} finally {
|
||||
TEST_VALUE_BAR_MAX_BYTES.qa = saved;
|
||||
}
|
||||
});
|
||||
|
||||
test('card fields clamp by UTF-8 bytes without splitting a character', () => {
|
||||
const long = '€'.repeat(100);
|
||||
const clamped = clampCardField(long);
|
||||
expect(Buffer.byteLength(clamped, 'utf8')).toBeLessThanOrEqual(CARD_FIELD_MAX_BYTES);
|
||||
expect(clamped.endsWith('...')).toBe(true);
|
||||
expect(clamped).not.toContain('\uFFFD');
|
||||
expect(clampCardField('short')).toBe('short');
|
||||
expect(renderValueCard({ protects: long, fails_when: 'x', why_new: 'y', seam: 'none' })).toStartWith(`Value: protects=${clamped}; fails_when=x;`);
|
||||
});
|
||||
});
|
||||
|
||||
describe('ship parent decision rules', () => {
|
||||
const rows = Object.fromEntries(shipGate.split('\n').filter(line => line.startsWith('| ') && !line.startsWith('| Step 7') && !line.startsWith('|---'))
|
||||
.map(line => line.split(' | ')).map(cells => [cells[0]!.slice(2), cells.slice(1).join(' | ')]));
|
||||
for (const [condition, uses, message] of [
|
||||
['Rating dispatch failed or timed out', 'skip the gate', MESSAGES.ratingUnavailable.message],
|
||||
['Zero paths, test-only diff', 'skip the gate', 'could not determine percentage'],
|
||||
['`Star rating: off`', '`coverage_pct`', 'weak paths still listed'],
|
||||
['`coverage_pct_value` missing, not a number, or outside 0..100', '`coverage_pct`', MESSAGES.valueCoverageUnavailable.message],
|
||||
['`coverage_pct_value` > `coverage_pct`', '`coverage_pct` (clamped)', MESSAGES.inconsistentCoverage.message],
|
||||
['Otherwise', '`coverage_pct_value`', ''],
|
||||
] as const) test(`${condition} -> ${uses}`, () => {
|
||||
const row = Object.entries(rows).find(([key]) => key.startsWith(condition));
|
||||
expect(row, `gate table lacks the row "${condition}"`).toBeDefined();
|
||||
expect(row![1]).toStartWith(uses);
|
||||
expect(row![1]).toContain(message);
|
||||
});
|
||||
|
||||
test('gate never substitutes 0 and parent processing tolerates old or malformed output', () => {
|
||||
expect(shipGate).toContain('never substitute 0');
|
||||
expect(shipSection).toContain('A missing new key counts\n as empty');
|
||||
expect(shipSection).toContain(MESSAGES.malformedKey.message);
|
||||
expect(shipSection).toContain(MESSAGES.allRejected.message);
|
||||
});
|
||||
|
||||
test('the rejection step removes new rejected files and reports tracked ones for a hunk revert', () => {
|
||||
const block = shipSection.slice(shipSection.indexOf(' while IFS= read -r f; do'), shipSection.indexOf(' REJECTED\n') + ' REJECTED\n'.length)
|
||||
.split('\n').map(line => line.replace(/^ {3}/, '')).join('\n');
|
||||
expect(block).toContain('<one rejected test path per line>');
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'test-value-rejected-'));
|
||||
try {
|
||||
const git = (...args: string[]) => spawnSync('git', args, { cwd: dir, encoding: 'utf8', timeout: 10_000 });
|
||||
git('init', '-q');
|
||||
fs.mkdirSync(path.join(dir, 'test'));
|
||||
fs.writeFileSync(path.join(dir, 'test/kept.test.ts'), 'test("a", () => {});\n');
|
||||
git('add', '.');
|
||||
git('-c', 'user.name=t', '-c', 'user.email=t@t', 'commit', '-q', '-m', 'seed');
|
||||
fs.appendFileSync(path.join(dir, 'test/kept.test.ts'), 'test("dup", () => {});\n');
|
||||
fs.writeFileSync(path.join(dir, 'test/new dup.test.ts'), 'test("dup", () => {});\n');
|
||||
const script = block.replace('<one rejected test path per line>', 'test/new dup.test.ts\ntest/kept.test.ts');
|
||||
const run = spawnSync('bash', ['-c', script], { cwd: dir, encoding: 'utf8', timeout: 10_000 });
|
||||
expect(run.status).toBe(0);
|
||||
expect(run.stdout).toContain('REMOVED: test/new dup.test.ts');
|
||||
expect(run.stdout).toContain('REVERT_HUNK: test/kept.test.ts');
|
||||
expect(fs.existsSync(path.join(dir, 'test/new dup.test.ts'))).toBe(false);
|
||||
expect(fs.existsSync(path.join(dir, 'test/kept.test.ts'))).toBe(true);
|
||||
} finally {
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test('PR body and Step 20 carry the value line, both numbers and the metrics', () => {
|
||||
const prBody = read('ship/sections/pr-body.md');
|
||||
expect(prBody).toContain('Coverage: {X}% value-weighted ({Y}% including {W} weakly covered paths)');
|
||||
expect(prBody).toContain('Test value: {K} tests written, {R} rejected by the authoring gate, {E} existing tests extended, {W} paths weakly covered (weak = ★, gate-failing or unrated).');
|
||||
expect(prBody).toContain('"1 test written", "1 existing test extended", "1 path weakly covered"');
|
||||
const ship = read('ship/SKILL.md');
|
||||
expect(ship).toContain('"coverage_pct":COVERAGE_PCT,"coverage_schema":2,"coverage_pct_value":COVERAGE_PCT_VALUE,"weak_gaps":WEAK_GAPS,"tests_extended":TESTS_EXTENDED,"tests_rejected":TESTS_REJECTED,"regression_proof":REGRESSION_PROOF');
|
||||
});
|
||||
});
|
||||
|
||||
describe('static review specialist stays in sync', () => {
|
||||
const specialist = read('review/specialists/testing.md');
|
||||
test('questions, catalog, evidence fields, codes, pragma and caller search', () => {
|
||||
const expected = [...QUESTIONS, ...CATALOG, ...REVIEW_EVIDENCE_FIELDS.map(field => `\`${field}\``), RETIREMENT_FIELDS.join(', '),
|
||||
...REASON_CODES.map(code => `\`${code}\``), `${PRAGMA} reason="<why>"`, CALLER_SEARCH_COMMAND, CALLER_SYMBOL_PATTERN, SWEEP_POINTER,
|
||||
'never an auto-delete', 'stays a coverage gap at its existing severity', 'one INFORMATIONAL line', 'Regression test without red proof',
|
||||
`test-value-bar.md#${MESSAGES.callerCheckUnavailable.anchor}`];
|
||||
const flat = specialist.replace(/\n/g, ' ').replace(/ +/g, ' ');
|
||||
expect(expected.filter(text => !flat.includes(text.replace(/\n/g, ' '))), `review/specialists/testing.md is out of sync with scripts/resolvers/test-value.ts`).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('template contracts', () => {
|
||||
test('consumers use their mode placeholder and both plan reviews state one contract', () => {
|
||||
expect(read('qa/SKILL.md.tmpl')).toContain('{{TEST_VALUE_BAR:qa}}');
|
||||
expect(read('qa-only/SKILL.md.tmpl')).toContain('{{TEST_VALUE_BAR:qa}}');
|
||||
expect(read('test-audit/SKILL.md.tmpl')).toContain('{{TEST_VALUE_BAR:audit}}');
|
||||
for (const skill of ['qa', 'qa-only']) expect(read(`${skill}/SKILL.md`), `${skill} lacks the rendered bar. ${FIX}`).toContain(bar('qa'));
|
||||
expect(read('test-audit/SKILL.md')).toContain(bar('audit'));
|
||||
const templates = spawnSync('git', ['ls-files', '*.tmpl'], { cwd: ROOT, encoding: 'utf8', timeout: 10_000 }).stdout.split('\n').filter(Boolean);
|
||||
expect(templates.filter(file => read(file).includes('prefer too many to too few'))).toEqual([]);
|
||||
expect(read('plan-ceo-review/SKILL.md.tmpl')).toContain('* Tests are required: every behavior tested; no test without a regression it would catch.');
|
||||
expect(read('plan-eng-review/SKILL.md.tmpl')).toContain('* **Tests:** every behavior tested; no test without a regression it would catch.');
|
||||
});
|
||||
|
||||
test('/test-audit is hard report-only when spawned', () => {
|
||||
const audit = read('test-audit/SKILL.md.tmpl').replace(/\s+/g, ' ');
|
||||
expect(audit).toContain('`SESSION_KIND: spawned` or `headless`, this run is hard report-only');
|
||||
expect(audit).toContain('treat every batch as C) stop');
|
||||
expect(audit).toContain('A) approve this batch B) skip it C) stop');
|
||||
});
|
||||
|
||||
test('the caller search keeps production resolvers and excludes tests', () => {
|
||||
const command = CALLER_SEARCH_COMMAND.replace('<symbol>', 'generateTestValueBar');
|
||||
const run = spawnSync('bash', ['-c', command], { cwd: ROOT, encoding: 'utf8', timeout: 30_000 });
|
||||
expect(run.status).toBe(0);
|
||||
const files = [...new Set(run.stdout.split('\n').filter(Boolean).map(line => line.split(':')[0]))];
|
||||
expect(files).toContain('scripts/resolvers/testing.ts');
|
||||
expect(files.filter(file => file!.startsWith('test/'))).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('docs and registration', () => {
|
||||
const docs = read('docs/test-value-bar.md');
|
||||
test('every degraded-mode message has a docs anchor and every rendered pointer resolves', () => {
|
||||
const anchors = new Set([...docs.matchAll(/^### ([a-z0-9-]+)$/gm)].map(match => match[1]));
|
||||
expect(Object.values(MESSAGES).map(({ anchor }) => anchor).filter(anchor => !anchors.has(anchor))).toEqual([]);
|
||||
const rendered = [shipSection, read('review/specialists/testing.md')].join('\n');
|
||||
const pointers = [...rendered.matchAll(/test-value-bar\.md#([a-z0-9-]+)/g)].map(match => match[1]);
|
||||
expect(pointers.length).toBeGreaterThan(5);
|
||||
expect(pointers.filter(anchor => !anchors.has(anchor!))).toEqual([]);
|
||||
});
|
||||
|
||||
test('/test-audit appears in every registration list', () => {
|
||||
const lists: [string, string][] = [
|
||||
['AGENTS.md', '| `/test-audit` |'],
|
||||
['README.md', '| `/test-audit` | **Test Auditor** |'],
|
||||
['README.md', '/deslop-shared-libs, /test-audit, /ship'],
|
||||
['docs/skills.md', '| [`/test-audit`](#test-audit) |'],
|
||||
['docs/skills.md', '## `/test-audit`'],
|
||||
['gstack/llms.txt', '[/test-audit](test-audit/SKILL.md)'],
|
||||
['SKILL.md.tmpl', 'invoke `/test-audit`'],
|
||||
['docs/PROJECT_STRUCTURE.md', 'test-audit/'],
|
||||
['test/fixtures/context-budget.json', '"test-audit":'],
|
||||
['test/helpers/touchfiles-data.ts', "'test-audit-report-only': ['test-audit/**'"],
|
||||
];
|
||||
expect(lists.filter(([file, text]) => !read(file).includes(text)).map(([file, text]) => `${file}: ${text}`)).toEqual([]);
|
||||
expect(read('README.md').split('/deslop-shared-libs, /test-audit, /ship').length - 1).toBe(2);
|
||||
});
|
||||
});
|
||||
Reference in new issue
Block a user