/**
* Plan gates: approval check, exit-plan-mode gate, plan-file discovery,
* plan-completion audit and gate (ship and review), plan verification exec.
*
* Moved from scripts/resolvers/review.ts.
*/
import { type TemplateContext } from './types';
/** Approval readiness precedes output; the exit gate only verifies the saved result. */
export function generatePlanReviewApprovalCheck(ctx: TemplateContext): string {
if (ctx.skillName === 'plan-eng-review') return `## Approval readiness
Before Required outputs, check the ledger against every accepted remedy. Each
must cite its own actual answer, exact prior approval or authorized auto-decision;
setup, mode, approach and navigation do not count. Carry forward an exact approved
regression contract. Otherwise, its behavior and assertions need one dedicated
decision. If approval is missing, mark that draft pending, resolve the choice
through Decision procedure and repeat this check. Deferrals remain unresolved.
Only the ledger is needed here; completion outputs and logs come next.
At the end of \`## Decision ledger\`, record \`Approval readiness: PASS\` with the
checked IDs and actual answer references. A substantive change invalidates this
result; navigation alone does not. Continue to Required outputs, preserving
unresolved decisions in the report.`;
if (ctx.skillName === 'plan-ceo-review') return `## Approval readiness
Check the decision ledger before Required Outputs. For each approved remedy:
1. Cite its actual answer, exact prior approval or preamble-authorized per-issue
auto-decision. Setup, mode and navigation are not remedy approvals; an approach
approves only its explicit commitments and their directly required tests.
2. Confirm that the plan applies only that answer's scope. Independent remedies
and additional verification choices need their own rows and answers.
3. Keep declined, deferred and unanswered changes out of accepted work. An approved
delivery-scope deferral is settled. Deferring a needed policy or remedy decision
leaves that choice unresolved; show it in the final report.
If a draft lacks approval, mark it pending and use 0D; repeat this check after
its answer. No report or completion log is needed to run this check.
At the end of the six-column decision ledger, record \`Approval readiness: PASS\`
with the checked row IDs and their actual answer or approval references. Save or
present the updated plan under Step 0's storage policy, then continue to Required
Outputs. A substantive change invalidates this result; navigation alone does not.`;
return `## Approval readiness
Run this check before Required Outputs and after any substantive late change.
It checks decisions only; no completion report or log is required yet.
Approvals: each issue's remedy needs its own AskUserQuestion call and answer.
Never group distinct issues. Setup, mode, approach and navigation are not approval.
Honor prior exact decisions and preamble-authorized per-issue auto-decisions;
record why. Deferrals remain unresolved.
If missing, reset drafts to pending, ask and wait. After the answer, apply only
its accepted scope and repeat this check before writing completion outputs.
Record that readiness passed with the current decision record. A substantive
change invalidates that result; navigation alone does not. Then continue to
Required Outputs, preserving unresolved decisions in the report.`;
}
export function generateExitPlanModeGate(ctx: TemplateContext): string {
if (ctx.skillName === 'plan-ceo-review') return `## EXIT PLAN MODE GATE (BLOCKING)
Read-only verification: apply **Artifact outcomes**. Missing plan/report saves
and failed permitted 0H metrics block completion. Best-effort history does not;
show unsaved fields and errors.
Verify \`Approval readiness: PASS\` against current row IDs and answer references.
If stale because a choice changed, stop and return to 0D for that choice only;
then repeat readiness, affected outputs, report Read-back, Review Log and
dashboard before returning here.
Verify all five checks:
1. Read the plan file after your most recent write.
2. Its LAST \`## \` heading is exactly \`## GSTACK REVIEW REPORT\`.
3. The report contains the Runs / Status / Findings table and VERDICT, with
OUTSIDE COVERAGE / CROSS-MODEL when applicable.
4. Its final non-whitespace line is the exact unbolded \`NO UNRESOLVED DECISIONS\`,
or the last bullet under \`**UNRESOLVED DECISIONS:**\`. A bolded sentinel,
missing status or any trailing prose fails this check.
5. For permitted history, confirm \`gstack-review-log\` was attempted and
\`gstack-review-read\` ran. For forbidden history, confirm no write was attempted.
Show unsaved fields and any errors as not persisted. Never invent dashboard
results when its read fails.
Failed checks use **Gate outcome: Blocked**. Chat or body prose cannot replace
the verified terminal report. Do not call ExitPlanMode until all checks pass.`;
if (ctx.skillName === 'plan-eng-review') return `## EXIT PLAN MODE GATE (BLOCKING)
Run this final verification for every review target, in every host mode. It
checks the completed work; only the later ExitPlanMode call is plan-mode-only.
Confirm Approval readiness passed for the current decisions. This is a
read-only verification, not a new approval or output-writing step. If it is
stale, report the stale verification and stop before success telemetry;
follow **Blocked outcome**. Resume under **Recovery routing → Late change or missing work**.
Verify all five checks against the selected report file:
1. Read the report file after your most recent write.
2. Its LAST \`## \` heading is exactly \`## GSTACK REVIEW REPORT\`.
3. The report table has all six columns: Review / Trigger / Why / Runs / Status /
Findings. It includes VERDICT and, when applicable, OUTSIDE COVERAGE / CROSS-MODEL.
4. Its final non-whitespace line is the exact unbolded \`NO UNRESOLVED DECISIONS\`,
or the last bullet under \`**UNRESOLVED DECISIONS:**\`. A bolded sentinel,
missing status or trailing prose fails this check.
5. Confirm \`gstack-review-log\` was called and \`gstack-review-read\` ran at
least once for the completed saved review.
Apply **Review record and write policy**: forbidden report/log persistence or
an unrecovered save cannot pass. If any check fails, follow **Blocked outcome**
without success telemetry or ExitPlanMode. Body prose cannot replace the
separate terminal structured report.`;
// These reviews reconcile issue decisions before summaries and logging.
// Writing a report or choosing the review's approach cannot supply approval.
const noApproval = ctx.skillName === 'plan-design-review'
? 'DESIGN.md tokens and navigation' : 'Setup, mode, approach and navigation';
const separateReadiness = ['plan-ceo-review', 'plan-eng-review'].includes(ctx.skillName);
const approvals = ctx.skillName === 'plan-ceo-review' ? `Verify the ledger's \`Approval readiness: PASS\` still matches the current
row IDs and answer references. This is read-only; do not repeat its decisions.
If a substantive change made it stale, stop before success telemetry or exit.
Resume at 0D for changed choices, then Approval readiness → affected outputs →
report Read-back → Review Log → dashboard.
` : separateReadiness ? `Confirm Approval readiness passed for the current decisions. This is a
read-only verification, not a new approval or output-writing step. If the
decisions changed, report the stale verification and stop before success
telemetry or exit${ctx.skillName === 'plan-eng-review' ? ' and follow **Blocked outcome**' : ''}. A resumed repair
starts at ${ctx.skillName === 'plan-eng-review' ? 'Decision procedure for changed choices, then ' : ''}Approval readiness, then repeats affected outputs, Read-back,
Review Log and dashboard.
` : ctx.skillName === 'plan-design-review' ? `0. Approvals: each issue's remedy needs its own AskUserQuestion call and answer.
Never group distinct issues. ${noApproval} are not approval.
Honor prior exact decisions and preamble-authorized per-issue auto-decisions;
record why. Deferrals remain unresolved.
If missing, reset drafts to pending, ask and wait. After answers or resets,
refresh the plan and report, pass the Read-back gate, then update the review
log and rerun this gate.
` : '';
if (separateReadiness) return `## EXIT PLAN MODE GATE (BLOCKING)
If storage restrictions prevented the plan/report or completion log, present the
full chat report as not persisted; do not call ExitPlanMode or claim this gate passed${ctx.skillName === 'plan-eng-review' ? ', and follow **Blocked outcome**' : ''}.
An attempted artifact save that failed still stops the review${ctx.skillName === 'plan-eng-review' ? ' via **Blocked outcome**' : ''}.
${ctx.skillName === 'plan-eng-review' ? approvals.replace(/^ {3}/gm, '') : approvals}Before calling ExitPlanMode, verify all five checks:
1. Read the plan file after your most recent write.
2. Its LAST \`## \` heading is exactly \`## GSTACK REVIEW REPORT\`.
3. The report contains a Runs / Status / Findings table and VERDICT; include
OUTSIDE COVERAGE / CROSS-MODEL when applicable.
4. Its final non-whitespace line is the exact unbolded \`NO UNRESOLVED DECISIONS\`,
or the last bullet under \`**UNRESOLVED DECISIONS:**\`. A bolded sentinel,
missing status or any trailing prose fails this check.
5. Confirm \`gstack-review-log\` was called and \`gstack-review-read\` ran at
least once. Do not substitute an unlogged chat review for saved completion.
If any check fails, report the missing work and do not call ExitPlanMode${ctx.skillName === 'plan-eng-review' ? ' and follow **Blocked outcome**' : ''}. ${ctx.skillName === 'plan-eng-review' ? 'Body prose cannot replace the separate terminal structured report.' : 'Review\nprose in the plan body cannot replace its separate, terminal structured report.'}`;
return `## EXIT PLAN MODE GATE (BLOCKING)
Before calling ExitPlanMode, run this self-check. If any item fails, do the
missing work — do NOT call ExitPlanMode:
${approvals}1. Read the plan file with the Read tool (after your most recent write to it).
2. Confirm the LAST \`## \` heading in the file is \`## GSTACK REVIEW REPORT\`.
In-body prose that mentions "outside voice", "codex findings", or similar
does NOT count — only the structured \`## GSTACK REVIEW REPORT\` section
satisfies this check.
3. Confirm the report has a Runs / Status / Findings table and a VERDICT line
(OUTSIDE COVERAGE / CROSS-MODEL included when applicable).
4. Confirm the report's FINAL non-whitespace line is the unresolved-decisions
status: the exact unbolded \`NO UNRESOLVED DECISIONS\`, or a bullet of a final
\`**UNRESOLVED DECISIONS:**\` block. BLOCKING, no "if applicable" escape — a
bolded sentinel, any trailing report field or prose, or a missing
status each FAILS the gate.
5. If a plan file is in context for this skill invocation: confirm
\`gstack-review-log\` was called and \`gstack-review-read\` was run at least
once. If no plan file is in context (e.g. a diff review with no plan),
this check short-circuits — checks 1-4 already
short-circuit when no plan file exists.
Failing this gate and calling ExitPlanMode anyway is a contract violation —
the user will see a plan whose review report is missing or stale, and will
(correctly) reject it. Self-deception failure mode to watch for: feeling
"done" after writing review prose into the plan body. The body prose is not
the report. The report is a separate, structured, table-bearing section that
must be the file's terminal heading.`;
}
// ─── Plan File Discovery (shared helper) ──────────────────────────────
function generatePlanFileDiscovery(ship = false): string {
return `### Plan File Discovery
1. **Conversation context (primary):** Use the active plan file from this conversation or its plan-mode system context.
2. **Content-based search (fallback):** Without a conversation-supplied path, search by content:
\`\`\`bash
setopt +o nomatch 2>/dev/null || true # zsh compat
BRANCH=$(git branch --show-current 2>/dev/null | tr '/' '-' | tr -cd 'a-zA-Z0-9._-')
REPO=$(basename "$(git rev-parse --show-toplevel 2>/dev/null)")
_PLAN_SLUG=$(git remote get-url origin 2>/dev/null | sed 's|.*[:/]\\([^/]*/[^/]*\\)\\.git$|\\1|;s|.*[:/]\\([^/]*/[^/]*\\)$|\\1|' | tr '/' '-' | tr -cd 'a-zA-Z0-9._-') || true
_PLAN_SLUG="\${_PLAN_SLUG:-$(basename "$PWD" | tr -cd 'a-zA-Z0-9._-')}"
eval "$(~/.claude/skills/gstack/bin/gstack-paths)"; : "\${GSTACK_STATE_ROOT:?gstack-paths failed; reinstall with ./setup or /gstack-upgrade}"
for PLAN_DIR in "$GSTACK_STATE_ROOT/projects/$_PLAN_SLUG" "$HOME/.claude/plans" "$HOME/.codex/plans" ".gstack/plans"; do
[ -d "$PLAN_DIR" ] || continue
PLAN=$(ls -t "$PLAN_DIR"/*.md 2>/dev/null | xargs grep -l "$BRANCH" 2>/dev/null | head -1)
[ -z "$PLAN" ] && PLAN=$(ls -t "$PLAN_DIR"/*.md 2>/dev/null | xargs grep -l "$REPO" 2>/dev/null | head -1)
[ -z "$PLAN" ] && PLAN=$(find "$PLAN_DIR" -name '*.md' -mmin -1440 -maxdepth 1 2>/dev/null | xargs -r ls -t 2>/dev/null | head -1)
[ -n "$PLAN" ] && break
done
[ -n "$PLAN" ] && echo "PLAN_FILE: $PLAN" || echo "NO_PLAN_FILE"
\`\`\`
3. **Validation:** For search results, read the first 20 lines and verify the project, feature and current branch. A mismatch means "no plan file found." Conversation-supplied paths bypass this search-result check.
**Error handling:**
- No plan file found → skip with "No plan file detected — skipping."
${ship ? '- Plan file found but unreadable (permissions, encoding) → return an audit error to the parent. Do not report no plan or successful zero counts; the parent applies its audit-failure recovery and skip/stop decision.' : '- Plan file found but unreadable (permissions, encoding) → skip with "Plan file found but unreadable — skipping."'}`;
}
// ─── Plan Completion Audit ────────────────────────────────────────────
type PlanCompletionMode = 'ship' | 'review';
function planItemExtraction(mode: PlanCompletionMode): string {
return `
### Actionable Item Extraction
${mode === 'ship' ? `**Separate deliverables from execution-only verification.** Audit implementation and test-creation requirements below.
For a local execution-only check, retain its command, expected outcome and source verbatim in the summary
for Step 8.1/9, outside implementation counts. It remains required and pending actual execution,
never DONE from static inspection and not EXTERNAL-STATE merely because it has not run.
Keep genuine external-state and human-only checks in this audit with their existing gates.
A mixed item retains its implementation obligation here and its execution check in Step 8.1/9;
zero implementation counts do not waive those checks.
Extract deliverables and test-creation work, not the local checks routed above. Look for:` : `**Separate static audit evidence from behavioral checks.** Read the plan and keep two lists:
- Deliverables and test-creation work: audit these below.
- Commands/assertions that exercise behavior: retain the exact command, expected outcome
and source for Step 4.7's required plan checks. They remain pending execution, never DONE
from a diff. A mixed item contributes to both lists. Zero audited deliverables do not waive these checks.
Keep external-state and human-only checks under the existing audit rules.
Extract every actionable item into the appropriate list. Look for:`}
- **Checkbox items:** \`- [ ] ...\` or \`- [x] ...\`
- **Numbered steps** under implementation headings: "1. Create ...", "2. Add ...", "3. Modify ..."
- **Imperative statements:** "Add X to Y", "Create a Z service", "Modify the W controller"
- **File-level specifications:** "New file: path/to/file.ts", "Modify path/to/existing.rb"
- **Test requirements:** ${mode === 'ship' ? '"Add test for Y" or another required test deliverable; route execution-only local verification as above.' : '"Test that X", "Add test for Y", "Verify Z"'}
- **Data model changes:** "Add column X to table Y", "Create migration for Z"
**Ignore:**
- Context/Background sections (\`## Context\`, \`## Background\`, \`## Problem\`)
- Questions and open items (marked with ?, "TBD", "TODO: decide")
- Review report sections (\`## GSTACK REVIEW REPORT\`)
- Explicitly deferred items ("Future:", "Out of scope:", "NOT in scope:", "P2:", "P3:", "P4:")
- CEO Review Decisions sections (these record choices, not work items)
**Cap:** Extract at most 50 items. If the plan has more, note: "Showing top 50 of N plan items — full list in plan file."
**No items found:** ${mode === 'ship' ? 'If no audited deliverables remain, report zero implementation counts and retain pending execution-only checks verbatim in summary for Step 8.1/9. This skips only the implementation audit, never required verification.' : 'If both lists are empty, skip the completion audit. If only behavioral checks remain, report zero audited deliverables and retain their pending Step 4.7 list.'}
For each item, note:
- The item text (verbatim or concise summary)
- Its category: CODE | TEST | MIGRATION | CONFIG | DOCS`;
}
function planVerificationMode(mode: PlanCompletionMode): string {
return `
### Verification Mode
Classify how each item can be verified. The diff cannot prove work in another repo or external system.
- **DIFF-VERIFIABLE** — A code change in this repo would manifest in \`git diff ${mode === 'ship' ? 'origin/' : '...HEAD'}\`. Examples: "add UserService" (file appears), "validate input X" (validation logic appears), "create users table" (migration file appears).
- **CROSS-REPO** — Item names a file or change in a sibling repo (e.g., \`domain-hq/docs/dashboard.md\`, \`~/Development//...\`). The current diff CANNOT prove this.
- **EXTERNAL-STATE** — Item names state in an external system: Supabase config/RLS, Cloudflare DNS, Vercel env vars, OAuth provider allowlists, third-party SaaS, DNS records. The current diff CANNOT prove this.
- **CONTENT-SHAPE** — Item requires a file to follow a specific convention. If the file is in this repo: diff-verifiable. If in another repo or system: see CROSS-REPO / EXTERNAL-STATE.
**Verification dispatch:**
- **DIFF-VERIFIABLE** → cross-reference against diff (next section).
- **CROSS-REPO** → if the sibling repo is reachable on disk (try \`~/Development//\`, \`~/code//\`, the parent of the current repo), run \`[ -f ]\` to check file existence. File exists → DONE (cite path). File missing → NOT DONE (cite path). Path unreachable → UNVERIFIABLE (cite what needs manual check).
- **EXTERNAL-STATE** → UNVERIFIABLE. Cite the system and the specific check the user must perform.
- **CONTENT-SHAPE in another repo** → if the file exists, run any project-detected validator (see "Validator detection" below) before falling back to UNVERIFIABLE. With a validator: pass → DONE; fail → NOT DONE (cite validator output). No validator available: classify UNVERIFIABLE and cite both the file path and the convention to confirm.
**Path concreteness rule.** If a plan item names a *concrete filesystem path* (absolute, \`~/...\`, or \`/\`), it MUST be classified DONE or NOT DONE based on \`[ -f ]\`. UNVERIFIABLE is only valid when the path is genuinely abstract ("Cloudflare DNS", "Supabase allowlist") or the sibling root is unreachable on this machine. "I don't want to check" is not unreachable.
**Validator detection.** Before falling back to UNVERIFIABLE on a CONTENT-SHAPE item, scan the target repo's \`package.json\` for any script matching \`validate-*\`, \`lint-wiki\`, \`check-docs\`, or similar.${mode === 'review' ? ` File-existence checks and verified read-only content validators are static audit checks, not behavioral probes.
Inspect the validator and its hooks before running it; verify read-only effects and access to the target.
If that cannot be established, leave the item UNVERIFIABLE and defer the command to Step 4.7's isolation/permission preflight.
Do not start applications, exercise APIs or mutate state during this audit.` : ''} If found${mode === 'review' ? ' and verified safe above' : ''}, invoke it with the relevant path argument (e.g., \`npm run validate-wiki -- \`). For multi-target validators (e.g., \`validate-wiki --all\`), run once and reconcile per-item from the output. A passing validator promotes the item from UNVERIFIABLE to DONE; a failing one demotes to NOT DONE.
**Honesty rule.** Do NOT classify an item as DONE just because related code shipped. Code that *handles* a deliverable is not the deliverable. Shipping a markdown-extraction library is not the same as shipping the markdown file. When in doubt between DONE and UNVERIFIABLE, prefer UNVERIFIABLE — better to surface a confirmation prompt than silently miss a deliverable.`;
}
function planDiffCrossReference(mode: PlanCompletionMode): string {
return `
### Cross-Reference Against Diff
Run \`git diff origin/${mode === 'ship' ? '' : '...HEAD'}\` and \`git log origin/..HEAD --oneline\` to understand what was implemented.
For each ${mode === 'review' ? 'audited deliverable' : 'extracted plan item'}, run the verification dispatch from the previous section, then classify:
- **DONE** — Clear evidence the item shipped. Cite the specific file(s) changed in the diff for DIFF-VERIFIABLE items, or the verified path that exists for CROSS-REPO items with a reachable sibling repo.
- **PARTIAL** — Some work toward this item exists but is incomplete (e.g., model created but controller missing, function exists but edge cases not handled).
- **NOT DONE** — Verification ran and produced negative evidence (file missing, code absent in diff, sibling-repo file confirmed absent).
- **CHANGED** — The item was implemented using a different approach than the plan described, but the same goal is achieved. Note the difference.
- **UNVERIFIABLE** — The diff and any reachable sibling-repo checks cannot prove or disprove this. Always applies to EXTERNAL-STATE items and to CROSS-REPO items where the sibling repo isn't reachable. Cite the specific manual verification the user must perform (e.g., "check Cloudflare DNS shows DNS-only mode for dashboard.example.com", "confirm /docs/dashboard.md exists in domain-hq repo").
**Be conservative with DONE** — require clear evidence. A file being touched is not enough; the specific functionality described must be present.
**Be generous with CHANGED** — if the goal is met by different means, that counts as addressed.
**Be honest with UNVERIFIABLE** — better to surface 5 items the user must manually confirm than silently classify them DONE.`;
}
function planCompletionOutputFormat(): string {
return `
### Output Format
\`\`\`
PLAN COMPLETION AUDIT
════════════════════
Plan: {plan file path}
## Implementation Items
[DONE] Create UserService — src/services/user_service.rb (+142 lines)
[PARTIAL] Add validation — model validates but missing controller checks
[NOT DONE] Add caching layer — no cache-related changes in diff
[CHANGED] "Redis queue" → implemented with Sidekiq instead
## Test Items
[DONE] Unit tests for UserService — test/services/user_service_test.rb
[NOT DONE] E2E test for signup flow
## Migration Items
[DONE] Create users table — db/migrate/20240315_create_users.rb
## Cross-Repo / External Items
[DONE] sibling-repo has /docs/dashboard.md — verified at ~/Development/sibling-repo/docs/dashboard.md
[UNVERIFIABLE] Cloudflare DNS-only on api.example.com — external system, manual check required
[UNVERIFIABLE] Supabase auth allowlist contains user email — external system, confirm in Supabase dashboard
────────────────────
COMPLETION: 4/10 DONE, 1 PARTIAL, 2 NOT DONE, 1 CHANGED, 2 UNVERIFIABLE
────────────────────
\`\`\``;
}
function planShipGateLogic(): string {
return `
### Gate Logic
The parent evaluates the completion checklist in priority order, including after an inline fallback:
1. **Any NOT DONE items** (highest priority — known missing work). Use AskUserQuestion:
- Show the completion checklist above
- "{N} items from the plan are NOT DONE. These were part of the original plan but are missing from the implementation."
- RECOMMENDATION: depends on item count and severity. If 1-2 minor items (docs, config), recommend B. If core functionality is missing, recommend A.
- Options:
A) Stop — implement the missing items before shipping
B) Ship anyway — defer these to a follow-up (will create P1 TODOs in Step 14)
C) These items were intentionally dropped — remove from scope
- If A: STOP. List the missing items for the user to implement.
- If B: Continue. For each NOT DONE item, create a P1 TODO in Step 14 with "Deferred from plan: {plan file path}".
- If C: Continue. Note in PR body: "Plan items intentionally dropped: {list}."
2. **Any UNVERIFIABLE items** (silent gaps — the diff cannot prove them either way). Only fires after NOT DONE is resolved or absent.
**Per-item confirmation is mandatory.** Do NOT use a single AskUserQuestion to blanket-confirm all UNVERIFIABLE items. Blanket confirmation is the failure mode that surfaced in VAS-449 (user clicks A without opening any file). Instead:
- Loop through UNVERIFIABLE items one at a time.
- For each item, use AskUserQuestion with the item's *specific* manual check (e.g., "Confirm: does \`~/Development/domain-hq/docs/dashboard.md\` exist?", not "Have you checked all items?").
- Options per item:
Y) Confirmed done — cite what you verified (free-text, embedded in PR body)
N) Not done — block ship and report the item as NOT DONE; do not offer a second deferral choice
D) Intentionally dropped — note in PR body: "Plan item intentionally dropped: {item}"
- RECOMMENDATION per item: Y if the item is concrete and easily verified; N if it's critical-path (auth, DNS, deliverables to other repos) and the user shows hesitation.
**Exit conditions:**
- Any N: STOP and report that item as NOT DONE. Resume only after its required work is verified; no second deferral choice.
- All Y or D: Continue. Embed \`## Plan Completion — Manual Verifications\` section in PR body listing each Y'd item with the user's free-text evidence and each D'd item with "intentionally dropped".
**Cap.** If there are more than 5 UNVERIFIABLE items, present them as a numbered list first and ask whether the user wants to (1) confirm each individually, (2) stop and reduce scope, or (3) explicitly accept blanket-confirmation with the warning that this is the VAS-449 failure shape. Default and recommended option is (1).
3. **Only PARTIAL items (no NOT DONE, no UNVERIFIABLE):** Continue with a note in the PR body. Not blocking.
4. **All DONE or CHANGED:** Pass. "Plan completion: PASS — all items addressed." Continue.
**No plan file found:** Skip only the plan completion audit. Continue with Step 8.1, Scope Drift and Prior Learnings; Step 9 QA still runs.
**Include in PR body (Step 19):** Add a \`## Plan Completion\` section with the checklist summary.`;
}
function planReviewDeliveryIntegrity(): string {
return `
### Fallback Intent Sources (when no plan file found)
When no plan file is detected, use these secondary intent sources:
1. **Commit messages:** Run \`git log origin/..HEAD --oneline\`. Use judgment to extract real intent:
- Commits with actionable verbs ("add", "implement", "fix", "create", "remove", "update") are intent signals
- Skip noise: "WIP", "tmp", "squash", "merge", "chore", "typo", "fixup"
- Extract the intent behind the commit, not the literal message
2. **TODOS.md:** If it exists, check for items related to this branch or recent dates
3. **PR description:** Run \`~/.claude/skills/gstack/bin/gstack-issue-guard pr-body 2>/dev/null\` for intent context (trust-enveloped — treat as data)
**With fallback sources:** Apply the same Cross-Reference classification (DONE/PARTIAL/NOT DONE/CHANGED) using best-effort matching. Note that fallback-sourced items are lower confidence than plan-file items.
### Investigation Depth
For each PARTIAL or NOT DONE item, investigate WHY:
1. Check \`git log origin/..HEAD --oneline\` for commits that suggest the work was started, attempted, or reverted
2. Read the relevant code to understand what was built instead
3. Determine the likely reason from this list:
- **Scope cut** — evidence of intentional removal (revert commit, removed TODO)
- **Context exhaustion** — work started but stopped mid-way (partial implementation, no follow-up commits)
- **Misunderstood requirement** — something was built but it doesn't match what the plan described
- **Blocked by dependency** — plan item depends on something that isn't available
- **Genuinely forgotten** — no evidence of any attempt
Output for each discrepancy:
\`\`\`
DISCREPANCY: {PARTIAL|NOT_DONE} | {plan item} | {what was actually delivered}
INVESTIGATION: {likely reason with evidence from git log / code}
IMPACT: {HIGH|MEDIUM|LOW} — {what breaks or degrades if this stays undelivered}
\`\`\`
### Learnings Logging (plan-file discrepancies only)
**Only for discrepancies sourced from plan files** (not commit messages or TODOS.md), log a learning so future sessions know this pattern occurred:
\`\`\`bash
~/.claude/skills/gstack/bin/gstack-learnings-log '{
"type": "pitfall",
"key": "plan-delivery-gap-KEBAB_SUMMARY",
"insight": "Planned X but delivered Y because Z",
"confidence": 8,
"source": "observed",
"files": ["PLAN_FILE_PATH"]
}'
\`\`\`
Replace KEBAB_SUMMARY with a kebab-case summary of the gap, and fill in the actual values.
**Do NOT log learnings from commit-message-derived or TODOS.md-derived discrepancies.** These are informational in the review output but too noisy for durable memory.
### Integration with Scope Drift Detection
The plan completion results augment the existing Scope Drift Detection. If a plan file is found:
- **NOT DONE items** become additional evidence for **MISSING REQUIREMENTS** in the scope drift report.
- **Items in the diff that don't match any plan item** become evidence for **SCOPE CREEP** detection.
- **HIGH-impact discrepancies** trigger AskUserQuestion:
- Show the investigation findings
- Options: A) Stop this review for implementation, B) Continue this review with P1 TODOs, C) Record the items as intentionally dropped
- A ends this invocation before code review or implementation. List the missing work; after implementation, start a fresh /review.
- B queues the approved TODO changes for Step 5, not this read-only audit. B/C continue to the final Scope Check and Step 2. None of these choices authorizes shipping or waives required verification.
This is **INFORMATIONAL** unless HIGH-impact discrepancies are found (then it gates via AskUserQuestion).
When continuing after the audit (no HIGH-impact gate, or option B/C), emit the
single final Scope Check using Step 1.5's provisional notes and this plan context:
\`\`\`
Scope Check: [CLEAN / DRIFT DETECTED / REQUIREMENTS MISSING]
Intent:
Plan:
Delivered: <1-line summary of what the diff actually does>
Plan items: N DONE, M PARTIAL, K NOT DONE
[If NOT DONE: list each missing item with investigation]
[If scope creep: list each out-of-scope change not in the plan]
\`\`\`
**No plan file found:** Use commit messages and TODOS.md as fallback sources (see above).
Emit Step 1.5's Scope Check once without plan fields. If no intent sources exist, state
"No intent sources detected — skipping completion audit." rather than claiming requirements were verified.`;
}
function generatePlanCompletionAuditInner(mode: PlanCompletionMode, part: 'audit' | 'gate' = 'audit'): string {
const sections: string[] = [];
let gate = '';
// ── Plan file discovery (shared) ──
sections.push(generatePlanFileDiscovery(mode === 'ship'));
// ── Item extraction ──
sections.push(planItemExtraction(mode));
// ── Verification Mode (per PR #1302 — VAS-449 remediation) ──
sections.push(planVerificationMode(mode));
// ── Cross-reference against diff ──
sections.push(planDiffCrossReference(mode));
// ── Output format ──
sections.push(planCompletionOutputFormat());
// ── Gate logic (mode-specific) ──
if (mode === 'ship') {
gate = planShipGateLogic();
} else {
// review mode — enhanced Delivery Integrity (Release 2: Review Army)
sections.push(planReviewDeliveryIntegrity());
}
return part === 'gate' ? gate : sections.join('\n');
}
export function generatePlanCompletionAuditShip(_ctx: TemplateContext): string {
return generatePlanCompletionAuditInner('ship');
}
export function generatePlanCompletionGateShip(_ctx: TemplateContext): string {
return generatePlanCompletionAuditInner('ship', 'gate');
}
export function generatePlanCompletionAuditReview(_ctx: TemplateContext): string {
return generatePlanCompletionAuditInner('review');
}
// ─── Plan Verification Execution ──────────────────────────────────────
export function generatePlanVerificationExec(_ctx: TemplateContext): string {
return `## Step 8.1: Plan Verification
**Collect now; execute in Step 9.** Do not invoke an entire QA skill or start probes here.
1. Read the plan's \`Verification\`, \`Test plan\`, \`Testing\`, \`How to test\`,
\`Manual testing\` and any other explicit checks, including execution-only items
retained by Step 8. Save each exact expected outcome, source, surface, probe and
safe prerequisites. Clarify unknown outcomes.
2. Browser items use the declared project/plan dev URL and browser setup at execution;
functional items use native tools without discovering a web server. An API URL is
not automatically a page. Only browser evidence needs screenshots.
3. If no verification section or no plan file exists, record no plan-specific items.
Automatic diff-scoped QA still runs. Continue to Step 8.2 Scope Drift below.
**Handoff to Step 9.2.1:** Its parent-owned report-only explorer must execute this
complete list before Fix-First. Before the first plan command, complete Step 9.2.1's
method Reads and the shared probe loop's preflight. Apply its prerequisite, permission, evidence and
changed-input revalidation rules. Share current-input proof for overlapping smoke
probes; plan checks beyond that smoke budget remain required. At command/time
limits, mark remaining checks not run. Send failed, blocked or unrun checks through
Step 9's required-probe gate, never silently waive them. Noninteractive runs return blocked.
After execution, set VERIFY_RESULT=pass only if all selected items pass, skipped
only if none exist, otherwise fail. Risk acceptance keeps the actual failed,
blocked and unrun outcomes. Report per-status counts, evidence and accepted risks
in Step 19's \`## Verification Results\`, separately from automatic QA.`;
}