/** * Native plan terminal and review-report detection. Moved from claude-pty-runner.ts. * Import through test/helpers/claude-pty-runner.ts from tests; pty/ modules import siblings directly. */ import * as fs from 'node:fs'; import * as path from 'node:path'; import { unresolvedPlanQuestionCalls, type NativePlanQuestionCall, type PlanCountTranscript } from '../plan-count-transcript'; import { isCurrentPlanApprovalScreen } from '../plan-count-pending-exit'; import { isRecordedDxManualNavigation } from '../dx-selected-navigation'; /** * Result of asserting that a plan file ends with `## GSTACK REVIEW REPORT` * as its last `## ` heading. `ok` is true iff the report is present AND no * other `## ` heading appears after it. Diagnostic fields are populated only * on failure to keep the success path cheap. */ export interface ReviewReportAtBottomResult { ok: boolean; reason?: string; trailingHeadings?: string[]; } /** * Assert that `## GSTACK REVIEW REPORT` is the last `## ` heading in a plan * file's content. Pure string operation — no filesystem access. Used by the * finding-count E2E tests as a second assertion on each test's produced plan. * * The plan-mode skill template mandates the agent move/append the review * report so it's always the last `##` section. A regression where the agent * appends additional sections after the report (or skips it entirely) ships * silently today; this assertion catches both. */ export function assertReviewReportAtBottom( content: string, ): ReviewReportAtBottomResult { const re = /^## GSTACK REVIEW REPORT\s*$/m; const match = re.exec(content); if (!match) { return { ok: false, reason: 'no GSTACK REVIEW REPORT section' }; } const after = content.slice(match.index + match[0].length); // Match any `## ` heading after the report. Reject `## ` followed by // newline-only (trailing-whitespace ## headers) to avoid false positives. const trailingHeadings = Array.from( after.matchAll(/^## \S.*$/gm), ).map((m) => m[0]); if (trailingHeadings.length > 0) { return { ok: false, reason: 'trailing ## heading(s) after GSTACK REVIEW REPORT', trailingHeadings, }; } return { ok: true }; } /** * A final native completion can replace a lost terminal heading, but never * an unanswered question, an old report, or a quoted completion example. * The expected path is supplied by the caller, not extracted for filesystem * access from model output. This does not add any question-count coverage. */ export function hasNativePlanCompletion( transcript: PlanCountTranscript, expectedPlanPath: string, startedAt: number, ): boolean { if (transcript.status !== 'ready' || !path.isAbsolute(expectedPlanPath) || !transcript.calls.length || transcript.calls.some(c => !c.answered || c.failed) || !transcript.assistantMessages.length) return false; const sessions = new Set([...transcript.calls.map(c => c.sessionId), ...transcript.assistantMessages.map(m => m.sessionId)]); if (sessions.size !== 1) return false; const messages = [...transcript.assistantMessages].sort((a, b) => Date.parse(a.timestamp) - Date.parse(b.timestamp)); const final = messages.at(-1)!; const finishedAt = Date.parse(final.timestamp); const answerTimes = transcript.calls.map(c => Date.parse(c.answeredAt ?? '')); if (!Number.isFinite(finishedAt) || !answerTimes.every(t => Number.isFinite(t) && t < finishedAt)) return false; const text = final.text.trim(); // Actual native prose, not ANSI aliases: the final announcement must say // the review is complete and identify the exact caller-owned deliverable. if (!/^(?:DX|Design|Eng(?:ineering)?|CEO) review complete[.!](?:\s|$)/i.test(text)) return false; const lastLine = text.split('\n').at(-1)!.trim(); if (lastLine !== `Plan written to: \`${expectedPlanPath}\`` && lastLine !== `Plan written to: ${expectedPlanPath}`) return false; const prose = text.replace(/```[\s\S]*?```|`[^`]*`/g, ''); if (/\?|\b(?:awaiting|waiting for|please (?:choose|answer|confirm))\b/i.test(prose)) return false; return hasCompletePlanReport(expectedPlanPath, Math.max(startedAt, ...answerTimes), finishedAt); } // Shared only by the opt-in Design report binding and its native completion // route. A quoted example is inert; an owned quoted status remains evidence. const DESIGN_CLOSURE_PROVISIONAL = /\b(?:example|sample|template|historical|previous|earlier|quoted|hypothetical|if|unless|until|once|assuming|provided|pending|would|will|could|might|may)\b/i; function designClosureText(text: string, expectedPlanPath: string): string { const state = /^(?:(?:still|now) )?(?:pending|failed|incomplete|unfinished|unresolved|open|withdrawn|superseded|cancelled|canceled|historical|not complete|not passed|not current)$/i; return text.replace(/\*\*/g, '') .replace(/`([^`\n]+)`/g, (_, value: string) => value === expectedPlanPath || state.test(value) ? value : '') .replace(/"[^"\n]*"|“[^”\n]*”|(? state.test(value.slice(1, -1)) ? value.slice(1, -1) : ''); } function conflictingDesignClosure(text: string): boolean { const owner = '(?:(?:the )?Design review(?: of PLAN\\.md)?|DESIGN CLEARED|(?:the )?(?:Design review )?exit gate|(?:the |this )?(?:review(?: report)?|report|gate|verdict|reviewed plan)|(?:(?:this|the|one|a|[1-9]\\d*) )?(?:design )?(?:decision|issue|finding)s?)'; return new RegExp(`(?:^|[.!?;]\\s+|\\n)(?:Correction:\\s*)?${owner} (?:(?:is|are|remains?|was|were|has been|have been) (?:(?:still|now) )?(?:not (?:the )?(?:complete|passed|current)|incomplete|unfinished|pending|failed|unresolved|open|withdrawn|superseded|cancelled|canceled|historical)|failed|did not pass|has not passed)\\b|${owner} (?:requires approval|applies only if approved)\\b|${owner}[^.!?\\n]*\\b(?:only if|conditional on|subject to)\\b|${owner} (?:belongs to|applies only to) (?:an? )?(?:another|different) (?:plan|project|review)\\b`, 'i').test(text) || new RegExp(`(?:^|[.!?;]\\s+|\\n)(?:If|When|Once|Unless|Assuming|Provided)\\b[^.!?\\n]*\\b${owner}\\b`, 'i').test(text); } export function hasCompletePlanReport(expectedPlanPath: string, minimumMtime: number, maximumMtime: number, allowRunHeaderForFailure = false, requiredReview?: 'Design'): boolean { if (!path.isAbsolute(expectedPlanPath)) return false; try { const stat = fs.lstatSync(expectedPlanPath); if (!stat.isFile() || stat.size > 2 * 1024 * 1024 || stat.mtimeMs < minimumMtime || stat.mtimeMs > maximumMtime) return false; // A template/example inside Markdown code is not the completed report. // Fences open with 3+ identical backticks or tildes, indented at most // three spaces. A close uses the same marker, at least the opening // length, and only horizontal whitespace after it. const reportLines: string[] = []; let fence: { marker: string; length: number } | undefined; for (const line of fs.readFileSync(expectedPlanPath, 'utf8').split(/\r?\n/)) { if (fence) { const close = /^ {0,3}(`{3,}|~{3,})[ \t]*$/.exec(line); if (close && close[1]![0] === fence.marker && close[1]!.length >= fence.length) fence = undefined; continue; } const open = /^ {0,3}(`{3,}|~{3,})(.*)$/.exec(line); // Backticks are forbidden in a backtick fence's info string. if (open && (open[1]![0] !== '`' || !open[2]!.includes('`'))) { fence = { marker: open[1]![0]!, length: open[1]!.length }; continue; } reportLines.push(line); } if (fence) return false; const content = reportLines.join('\n'); if (!assertReviewReportAtBottom(content).ok) return false; const report = content.slice(content.indexOf('## GSTACK REVIEW REPORT')); // Reject a provisional heading; retain both clean and issues-open reports. const rows = report.split('\n'); // Positive completion keeps the canonical Review header. A failure-only // diagnostic may also recognize the Run header found in a native report; // it still requires a complete, fresh caller-owned deliverable. const tableHeader = allowRunHeaderForFailure ? /^\|(?:[^\n|]*\|)*[ \t]*(?:Review|Run)[ \t]*\|/ : /^\|[^\n]*Review[^\n]*\|/; const table = rows.findIndex(line => tableHeader.test(line)); const completeTable = table >= 0 && /^\|[ :|-]+\|$/.test(rows[table + 1] ?? '') && /^\|.*[a-zA-Z].*\|$/.test(rows[table + 2] ?? ''); const decisions = rows.findIndex(line => /^\*\*UNRESOLVED DECISIONS:\*\*$/.test(line)); const trailing = rows.slice(decisions + 1).filter(line => line.trim()); const closed = rows.filter(line => line.trim()).at(-1) === 'NO UNRESOLVED DECISIONS' || (decisions >= 0 && trailing.length > 0 && trailing.every(line => /^[-*] \S|^\+ \d+ unresolved from prior reviews$/.test(line))); if (requiredReview === 'Design') { const cells = (row: string) => row.split('|').slice(1, -1).map(cell => cell.trim()); const header = cells(rows[table] ?? ''); const design = rows.slice(table + 2).filter(row => row.startsWith('|')) .map(cells).filter(row => row[header.indexOf('Review')] === 'Design Review'); const verdicts = rows.filter(row => /^(?:[-*] )?(?:\*\*)?VERDICT:/i.test(row)); const verdict = designClosureText(verdicts[0] ?? '', expectedPlanPath).replace(/^(?:[-*] )?VERDICT:\s*/i, ''); if (design.length !== 1 || !/^(?:clean|clear(?: \(full\))?|complete[d]?)$/i.test(design[0]![header.indexOf('Status')] ?? '') || verdicts.length !== 1 || !/^DESIGN CLEARED\b/i.test(verdict) || /\?/.test(verdict) || /^DESIGN CLEARED\s+(?:is|was|were|has been|had been|not|never|no longer)\b/i.test(verdict) || DESIGN_CLOSURE_PROVISIONAL.test(verdict) || conflictingDesignClosure(verdict) || conflictingDesignClosure(designClosureText(report, expectedPlanPath)) || rows.filter(row => row.trim()).at(-1) !== 'NO UNRESOLVED DECISIONS') return false; } return completeTable && /^(?:[-*] )?(?:\*\*)?VERDICT:(?:\*\*)?[ \t]*[A-Za-z]/m.test(report) && closed; } catch { return false; } } /** A final owned gate with no review questions is missing coverage, never success. * The caller may identify completed setup/navigation calls; unclassified, * pending, or failed questions cannot be dismissed by this failure diagnostic. */ export function isQuestionlessNativePlanExit( transcript: PlanCountTranscript, expectedPlanPath: string, startedAt: number, screen: string, nonReviewCalls: ReadonlySet = new Set(), ): boolean { if (!isCurrentPlanApprovalScreen(screen)) return false; if (transcript.status !== 'ready' || transcript.calls.some(call => !call.sessionId || !call.toolUseId || !call.answered || call.failed || !nonReviewCalls.has(`${call.sessionId}:${call.toolUseId}`) || !call.questions.length || call.questions.some(q => !call.answers?.[q.question]) || !Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length)) return false; const ready = [...(transcript.planReadyRequests ?? [])] .sort((a, b) => Date.parse(a.timestamp) - Date.parse(b.timestamp)).at(-1); if (!ready || ready.failed || !ready.sessionId || !ready.toolUseId) return false; const at = Date.parse(ready.timestamp); const sessions = new Set([...transcript.calls.map(call => call.sessionId), ...transcript.assistantMessages.map(m => m.sessionId), ...(transcript.planReadyRequests ?? []).map(r => r.sessionId)]); return sessions.size === 1 && Number.isFinite(at) && at >= startedAt && at <= Date.now() && transcript.calls.every(call => { const answerAt = Date.parse(call.answeredAt ?? ''); return Number.isFinite(answerAt) && answerAt >= startedAt && answerAt < at; }) && hasCompletePlanReport(expectedPlanPath, startedAt, at, true); } export interface NativePlanTerminalReview { transcript: PlanCountTranscript; report: string; reportMtimeMs: number; startedAt: number; finishedAt: number; deadlineAt: number; } export interface NativePlanTerminalAssessment { administrativeCallIds: readonly string[]; substantiveCallIds: readonly string[]; } export type NativePlanTerminalEvaluator = (input: NativePlanTerminalReview) => Promise; /** The opt-in semantic assessment runs once at a real owned Exit, before the * existing freshness gate. It may exclude only native calls it has assessed; * the unchanged terminal check then requires a later report for all other ACKs. */ export async function evaluateOwnedNativePlanTerminal(transcript: PlanCountTranscript, expectedPlanPath: string, startedAt: number, deadlineAt: number, evaluate: NativePlanTerminalEvaluator): Promise<{ administrative: ReadonlySet; substantive: ReadonlySet } | undefined> { const finishedAt = Date.now(); if (!Number.isFinite(deadlineAt) || finishedAt >= deadlineAt) throw new Error('Native terminal assessment: absolute case deadline exhausted'); const snapshot = structuredClone(transcript), calls = snapshot.calls; const identities = calls.map(call => `${call.sessionId}:${call.toolUseId}`); const sessions = new Set([...calls.map(call => call.sessionId), ...snapshot.assistantMessages.map(m => m.sessionId), ...(snapshot.planReadyRequests ?? []).map(r => r.sessionId)]); const ready = [...(snapshot.planReadyRequests ?? [])].sort((a,b) => Date.parse(a.timestamp) - Date.parse(b.timestamp)).at(-1); if (snapshot.status !== 'ready' || !calls.length || sessions.size !== 1 || new Set(identities).size !== calls.length || !ready?.sessionId || !ready.toolUseId || ready.failed !== false || identities.includes(`${ready.sessionId}:${ready.toolUseId}`) || !Number.isFinite(Date.parse(ready.timestamp)) || Date.parse(ready.timestamp) <= startedAt || Date.parse(ready.timestamp) > finishedAt || calls.some(call => !call.sessionId || !call.toolUseId || call.answered !== true || call.failed !== false || !Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length || !call.questions.length || Object.keys(call.answers ?? {}).length !== call.questions.length || !Number.isFinite(Date.parse(call.answeredAt ?? '')) || Date.parse(call.answeredAt!) < startedAt || Date.parse(call.answeredAt!) >= Date.parse(ready.timestamp) || call.questions.some(q => q.multiSelect || new Set(q.options.map(o => o.label)).size !== q.options.length || !q.options.some(o => o.label === call.answers?.[q.question])))) return undefined; if (!hasCompletePlanReport(expectedPlanPath, startedAt, Date.parse(ready.timestamp))) return undefined; const stat = fs.lstatSync(expectedPlanPath), report = fs.readFileSync(expectedPlanPath, 'utf8'); let timer: ReturnType | undefined; try { const result = await Promise.race([evaluate({ transcript: structuredClone(snapshot), report, reportMtimeMs: stat.mtimeMs, startedAt, finishedAt, deadlineAt }), new Promise((_, reject) => { timer = setTimeout(() => reject(new Error('Native terminal assessment: absolute case deadline exhausted')), Math.max(1, deadlineAt - Date.now())); })]); if (Date.now() >= deadlineAt) throw new Error('Native terminal assessment: absolute case deadline exhausted'); const validIds = (ids: readonly string[]) => Array.isArray(ids) && new Set(ids).size === ids.length && ids.every(id => identities.includes(id)); if (!result || !validIds(result.administrativeCallIds) || !validIds(result.substantiveCallIds) || !result.substantiveCallIds.length || result.administrativeCallIds.some(id => result.substantiveCallIds.includes(id))) throw new Error('Native terminal assessment returned foreign, duplicate, overlapping or empty substantive identities'); const after = fs.lstatSync(expectedPlanPath); if (!after.isFile() || after.dev !== stat.dev || after.ino !== stat.ino || after.size !== stat.size || after.mtimeMs !== stat.mtimeMs || fs.readFileSync(expectedPlanPath, 'utf8') !== report) throw new Error('Native terminal report changed during assessment'); const administrative = new Set(result.administrativeCallIds); if (!hasNativePlanTerminal(snapshot, expectedPlanPath, startedAt, 'plan_ready', administrative)) throw new Error('Native terminal report is not fresh after every substantive native answer'); return { administrative, substantive: new Set(result.substantiveCallIds) }; } finally { clearTimeout(timer); } } /** Native report completion is independent of the terminal's streamed headings. */ export function hasNativePlanTerminal( transcript: PlanCountTranscript, expectedPlanPath: string, startedAt: number, frame: 'completion_summary' | 'plan_ready', administrativeCalls: ReadonlySet = new Set(), ): boolean { if (transcript.status !== 'ready' || !transcript.calls.length || transcript.calls.some(c => (!c.answered && !c.failed) || (c.answered && c.failed)) || unresolvedPlanQuestionCalls(transcript.calls).length) return false; const sessions = new Set([...transcript.calls.map(c => c.sessionId), ...transcript.assistantMessages.map(m => m.sessionId), ...(transcript.planReadyRequests ?? []).map(r => r.sessionId)]); if (sessions.size !== 1) return false; const answered = transcript.calls.filter(c => c.answered); if (!answered.length) return false; const answerTimes = answered.map(c => Date.parse(c.answeredAt ?? '')); if (!answerTimes.every(Number.isFinite)) return false; const latestAnswer = Math.max(...answerTimes); const latestReady = [...(transcript.planReadyRequests ?? [])] .sort((a, b) => Date.parse(a.timestamp) - Date.parse(b.timestamp)).at(-1); if (latestReady?.failed) return false; const modifyingAnswers = answered.filter(c => !administrativeCalls.has(`${c.sessionId}:${c.toolUseId}`) && !isCompletedDxHandoff(c)) .map(c => Date.parse(c.answeredAt!)); if (!modifyingAnswers.length || !hasCompletePlanReport(expectedPlanPath, Math.max(startedAt, ...modifyingAnswers), Date.now())) return false; if (frame === 'plan_ready') { // A real pending ExitPlanMode call is the approval gate. Do not answer // it or mistake its failed tool result for a completed review. return Boolean(latestReady && Date.parse(latestReady.timestamp) > latestAnswer && Date.parse(latestReady.timestamp) <= Date.now()); } const final = [...transcript.assistantMessages] .sort((a, b) => Date.parse(a.timestamp) - Date.parse(b.timestamp)).at(-1); if (!final || !Number.isFinite(Date.parse(final.timestamp)) || Date.parse(final.timestamp) <= latestAnswer) return false; const text = final.text.trim(); // Only finished, fixture-scoped native prose supplies this evidence. A // rendered heading, quoted example, proposed write, or pending question // cannot substitute for an actual completion message. if (/^(?:>|`{3,}|~{3,}|(?:example|sample|template|quoted)\b)/i.test(text) || /\b(?:cannot|can't|unable to)\s+(?:complete|finish)|\b(?:awaiting|waiting for|please (?:choose|answer|confirm))\b/i.test(text) || text.endsWith('?')) return false; if (/^(?:CEO|Eng(?:ineering)?|Design|DX) review complete[.!](?:\s|$)/i.test(text)) return true; // A full native summary can precede the final report write; the independent // fresh-file check above prevents the observed heading-before-Write race. const lines: string[] = []; let fence: { marker: string; length: number } | undefined; for (const line of text.split(/\r?\n/)) { const delimiter = /^ {0,3}(`{3,}|~{3,})(.*)$/.exec(line); if (delimiter) { if (!fence) fence = { marker: delimiter[1]![0]!, length: delimiter[1]!.length }; else if (delimiter[1]![0] === fence.marker && delimiter[1]!.length >= fence.length && !delimiter[2]!.trim()) fence = undefined; continue; } if (!fence && !/^(?: {4}|\t| {0,3}>)/.test(line)) lines.push(line); } if (fence) return false; // A Design review may finish at the user's manual handoff without invoking // ExitPlanMode. Bind its affirmative review and passed gate to this report; // the common native ownership, answered-call and fresh-file checks above // still apply. Other skills retain their existing completion routes below. const designText = designClosureText(lines.join('\n'), expectedPlanPath); const designParagraphs = designText.split(/\n\s*\n/).map(p => p.trim()); const designComplete = /(?:^|[.!]\s+)(?:The )?Design review(?: of PLAN\.md)? (?:is complete|has been completed)[.!](?:\s|$)/i; const designGate = /(?:^|[.!]\s+)(?:The )?(?:Design review )?exit gate (?:has )?passed[.:!](?:\s|$)/i; const escapedPlanPath = expectedPlanPath.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); const designReport = new RegExp(`(?:^|[.!]\\s+)(?:The )?(?:reviewed plan|review report) (?:is at|was written to|is saved at) ${escapedPlanPath}[.!](?:\\s|$)`, 'i'); const designCompleted = designParagraphs.filter(p => designComplete.test(p) && designReport.test(p)); const designPassed = designParagraphs.filter(p => designGate.test(p)); const sourcedDesign = (paragraph: string) => DESIGN_CLOSURE_PROVISIONAL.test(paragraph) || /\b(?:example|sample|template|historical|previous|earlier|quote|source|emit|print)\b.*[::]\s*$/i .test(designParagraphs[designParagraphs.indexOf(paragraph) - 1] ?? ''); if (designCompleted.length === 1 && designPassed.length === 1 && (designText.match(/\b(?:reviewed plan|review report) (?:is at|was written to|is saved at)\b/gi)?.length ?? 0) === 1 && ![...designCompleted, ...designPassed].some(sourcedDesign) && !conflictingDesignClosure(designText) && Date.parse(final.timestamp) <= Date.now() && hasCompletePlanReport(expectedPlanPath, Math.max(startedAt, ...modifyingAnswers), Date.parse(final.timestamp), false, 'Design')) return true; // Typed Design completion owns a current status and saved artifact. Markdown // headings/field emphasis and explanatory prose are presentation, not a // second approval or a substitute for the strict native/report checks. const completionHeadings = lines.map((line, index) => /^(?:#{1,6}\s+)?(?:\*\*)?(?:Completion(?:\s+(?:summary|report))?|(?:Design\s+)?Review\s+(?:complete|completion(?:\s+summary)?))(?:\*\*)?:?\s*$/i.test(line.replace(/\*\*/g, '').trim()) ? index : -1) .filter(index => index >= 0); if (completionHeadings.length === 1) { const start = completionHeadings[0]!; const preceding = lines.slice(0, start).filter(line => line.trim()).at(-1) ?? ''; const section = lines.slice(start + 1); const plain = section.map(line => line.replace(/\*\*/g, '').trim()); const status = plain.filter(line => /^(?:[-*]\s+)?STATUS:/i.test(line)); const done = /^(?:[-*]\s+)?STATUS:\s*DONE(?:\s+[—–:-]\s+(.+)|[.!](?:\s+(.*))?)?\s*$/i.exec(status[0] ?? ''); // A punctuated status may close with resolved-decision/read-only facts. // These clauses establish no new permission, next action or plan-mode exit. const harmlessClosure = !done?.[2] || done[2].split(/[.;!]\s*/).filter(Boolean).every(clause => /^(?:No (?:unresolved|open|pending) (?:design )?(?:decisions|issues|findings)|(?:(?:I am|We are) )?(?:Staying|Remaining) in plan mode|Nothing outside (?:the )?(?:plan|report) file (?:was|has been) (?:edited|changed|modified)|No implementation (?:was|has been) (?:started|performed)|Implementation (?:has not started|was not started))$/i.test(clause.trim())); const pathFields = plain.filter(line => /^(?:[-*]\s+)?(?:Plan (?:written|saved)(?: to\b|:)|(?:What changed|Plan|Report|Output|Artifact):)/i.test(line)); const saved = pathFields.filter(line => { const value = line.replace(/^[-*]\s+/, '').replace(/^Plan (written|saved):\s*/i, 'Plan $1 to ').replace(/^(?:What changed|Plan|Report|Output|Artifact):\s*/i, '').trim(); if (DESIGN_CLOSURE_PROVISIONAL.test(value) || /^(?:[>"“'‘]|`{3}|~{3})/.test(value)) return false; // The expected absolute path or its exact basename identifies this one // caller-owned report. Reject foreign/ambiguous paths before stripping // Markdown; a filename hidden in quoted prose supplies no authority. const paths = [...value.matchAll(/[^\s`"'<>()[\]{};,]+\.md(?=$|[\s`"'.,;:)])/g)].map(m => m[0]); if (paths.length !== 1 || ![expectedPlanPath, path.basename(expectedPlanPath)].includes(paths[0]!)) return false; const current = value.replace(/`([^`\n]+)`/g, (_, v: string) => paths.includes(v) ? v : '') .replace(/"[^"\n]*"|“[^”\n]*”|(? /^(?:(?:review )?report is (?:the )?(?:last|final) (?:section|heading)|saved|written|read[- ]?back verified|verified by read[- ]?back)$/i.test(clause.trim())); return new RegExp(`^Plan (?:written|saved) to ${token}(?:[.!](?:\\s|$)|\\s|$)`, 'i').test(current) || Boolean(verifiedAnnotation) || new RegExp(`^${token}\\s+(?:now\\s+)?(?:contains|carries|includes|records)\\s+`, 'i').test(current) && /\b(?:review report|reviewed plan)\b/i.test(current); }); // A typed completion's status cannot substitute for the actual answers. // Failed calls retain the shared later-answer resolution rule above. const ownedAnswers = answered.every(call => call.sessionId && call.toolUseId && call.questions.length > 0 && new Set(call.questions.map(q => q.question)).size === call.questions.length && Object.keys(call.answers ?? {}).length === call.questions.length && Array.isArray(call.unansweredQuestionIndices) && call.unansweredQuestionIndices.length === 0 && call.questions.every(q => q.options.some(o => o.label === call.answers?.[q.question]))); if (ownedAnswers && !section.some(line => /^#{1,6}\s/.test(line)) && !/\b(?:example|sample|template|historical|previous|earlier|quote|source|emit|print)\b.*[::]\s*$/i.test(preceding) && status.length === 1 && done && harmlessClosure && !DESIGN_CLOSURE_PROVISIONAL.test(done[1] ?? '') && pathFields.length === 1 && saved.length === 1 && !conflictingDesignClosure(designText) && Date.parse(final.timestamp) <= Date.now() && hasCompletePlanReport(expectedPlanPath, Math.max(startedAt, ...modifyingAnswers), Date.parse(final.timestamp), false, 'Design')) return true; } const summary = lines.findIndex(line => /^(?:#{1,6}\s*)?(?:\*\*)?Completion\s+summary(?:\*\*)?\s*:?[ \t]*$/i.test(line.trim())); const preceding = lines.slice(0, summary).filter(line => line.trim()).at(-1) ?? ''; if (summary < 0 || /\b(?:example|sample|template|quote|emit|print)\b.*[::]\s*$/i.test(preceding)) return false; const rows = lines.slice(summary + 1).filter(line => /^\s*(?:[-*] |\|)/.test(line)); return rows.length > 0 && rows.some(line => /\b(?:review|issues?|findings?|gaps?|resolved|scope)\b/i.test(line)); } /** G's post-report DX next-step menu changes orchestration, not plan decisions. */ function isCompletedDxHandoff(call: NativePlanQuestionCall): boolean { if (!call.answered || call.failed || call.questions.length !== 1 || !Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length) return false; const q = call.questions[0]!; if (isRecordedDxManualNavigation(call) || manualDxTaskNavigation(call) || resolvedDxTaskNavigation(call) || specifiedDxTaskNavigation(call) || architecturalDxTaskNavigation(call)) return true; const header = q.header.trim().replace(/^D\s*\d+\s*(?:[—–:-]\s*)?/i, ''); const question = q.question.replace(/^D\s*\d+\s*[—–:-]\s*/i, ''); const completionLines: string[] = []; let completionFence: { marker: string; length: number } | undefined; for (const line of question.split('\n')) { const delimiter = /^ {0,3}(`{3,}|~{3,})(.*)$/.exec(line); if (delimiter) { if (!completionFence) completionFence = { marker: delimiter[1]![0]!, length: delimiter[1]!.length }; else if (delimiter[1]![0] === completionFence.marker && delimiter[1]!.length >= completionFence.length && !delimiter[2]!.trim()) completionFence = undefined; } else if (!completionFence && !/^(?: {4}|\t|\s*>)/.test(line)) completionLines.push(line); } const completionProse = completionLines.join('\n'); const closedReviewNavigation = /^What(?:['’]s)? next\?\s*\n/i.test(question) && /(?:^|\n)(?:ELI10:\s*)?(?:The )?DX review (?:is )?(?:done|complete)[.!](?:\s|$)/i.test(completionProse); if (q.multiSelect || !/^next(?:\s+steps?)?$/i.test(header) || (!closedReviewNavigation && !/^DX review (?:is )?(?:done|complete)[.!]/i.test(question))) return false; const ids = [...q.question.matchAll(/]+)>/gi)]; if ((q.question.match(/ o.description ?? '')].join('\n'); const state = '(?:(?:is|are|was|were|becomes?|became)|(?:will|would|can|could|may|might) (?:be|become))'; const closure = `(?:the )?(?:DX review (?:${state} )?(?:done|complete)|(?:all )?(?:decisions?|gaps?|issues?|findings?) (?:${state} )?resolved)`; const conditionalClosure = new RegExp(`\\b(?:once|when|after)\\b[^.!?\\n]*${closure}|${closure}[^.!?\\n]*\\b(?:once|when|after)\\b`, 'i'); if ((q.question.match(/\?/g)?.length ?? 0) !== 1 || q.options.some(option => /\?/.test(option.description ?? '')) || !/\brequired (?:eng review )?gate (?:for|before) shipping\b/i.test(context) || /\b(?:unresolved|pending|remaining|outstanding|if|unless|until)\b|\b(?:gap|issue|finding|decision)s?\s+(?:still\s+)?remains?\b/i.test(context) || /\b(?:not all|not (?:done|complete|resolved)|only after)\b/i.test(context) || conditionalClosure.test(context) || /(?:^|[.!?;]\s+|\b(?:proceed to|continue to|should|must|will|can|could|would|may|need to)\s+)(?:(?:please|first|then|also)\s+)*(?:add|fix|package|implement|resolve|decide)\b/im.test(context)) return false; } } else { // Some native final menus omit a qid. Require an explicit finished-review // declaration plus resolved findings and the navigation-only question; // malformed/unknown identities and outstanding decisions remain blockers. const navigation = /\bWhat(?:['’]s)? next\?/i.exec(q.question); const afterQuestion = navigation ? q.question.slice(navigation.index + navigation[0].length).trim() : ''; resultRecap = /^DX Review result:\s*\d+(?:\.\d+)?\/10\s*(?:→|->)\s*\d+(?:\.\d+)?\/10[.!]/i.test(afterQuestion) && /(?:^|\n)Recommendation:\s*(?:[A-Z]\s*[—–-]\s*)?\/plan-eng-review[.!](?:\s|$)/i.test(afterQuestion); const context = [q.question, ...q.options.map(o => o.description ?? '')].join('\n'); if (resultRecap && (/\b(?:if|unless|until)\b|\bnot(?:\s+[a-z-]+){0,4}\s+resolved\b/i.test(context) || q.options.some(option => /\?/.test(option.description ?? '')) || !/\brequired gate before shipping\b/i.test(context))) return false; if (/ { const label = o.label.trim().replace(/\s*\(recommended\)\s*$/i, ''); return closedReviewNavigation ? label.replace(/^[A-Z][.)]\s*/i, '') : label; }); const runEng = (label: string) => /^Run \/plan-eng-review(?: next)?$/i.test(label); const ready = (label: string) => /^Ready to implement(?:\s*[—–-]\s*run \/devex-review after shipping)?$/i.test(label) || (resultRecap && /^Start implementing now$/i.test(label)); const manual = (label: string) => /^Skip(?:, handle manually|\s*[—–-]\s*I['’]ll handle next steps manually)$/i.test(label) || (resultRecap && /^Skip, handle next steps manually$/i.test(label)); return labels.every(label => runEng(label) || ready(label) || manual(label)) && labels.filter(runEng).length === 1 && labels.filter(manual).length === 1 && q.options.some(o => o.label === call.answers?.[q.question]); } /** A selected manual handoff leaves already recorded DX decisions unchanged. */ function manualDxTaskNavigation(call: NativePlanQuestionCall): boolean { const q = call.questions[0]!; if (call.answered !== true || call.failed !== false || !call.sessionId || !call.toolUseId || !Number.isFinite(Date.parse(call.answeredAt ?? '')) || q.multiSelect || q.header.trim() !== 'Next steps' || q.options.length !== 3 || new Set(q.options.map(o => o.label)).size !== 3 || Object.keys(call.answers ?? {}).length !== 1) return false; const prose = (text: string) => text.split(/\r?\n/).filter(line => !/^\s*>/.test(line) && !/^\s*(["'`]).*\1\s*$/.test(line)).join('\n').trim(); const text = prose(q.question); if (/```|~~~||→) (?:10|[0-9])(?:\.\d+)?\/10\))?\. (?:What should happen next|What happens next|What['’]s next)\?\n/i.test(text) || !/(?:^|\n)ELI10:\s*The DX review (?:found|identified)\b/i.test(text) || !/(?:^|[.!]\s+)(?:All (?:are|have been) (?:written|recorded) (?:in|into) the plan as tasks\b|All DX decisions and tasks are recorded in the plan\.)/i.test(text)) return false; const metadata = /(?:^|\n)Project\/branch\/task:\s*([^\n]*)/i.exec(text)?.[1]; if (metadata && /^(?:if|unless|when|once|after|provided|proposed|optional|source|example|historical|earlier review|previously)\b/i.test(metadata)) return false; const label = (value: string) => value.trim().replace(/\s*\((?:recommended|required gate)\)\s*$/i, ''); const manual = (value: string) => /^Skip(?:,|\s*[—–-])\s*I['’]ll handle next steps manually$/i.test(label(value)); const eng = (value: string) => /^Run \/plan-eng-review next$/i.test(label(value)); const implement = (value: string) => /^Ready to implement(?:,|\s*[—–-])\s*run \/devex-review after shipping$/i.test(label(value)); if (q.options.filter(o => manual(o.label)).length !== 1 || q.options.filter(o => eng(o.label)).length !== 1 || q.options.filter(o => implement(o.label)).length !== 1) return false; const selected = q.options.find(o => o.label === call.answers?.[q.question]); if (!selected || !manual(selected.label)) return false; const description = prose(selected.description ?? '').replace(/[✅❌]/g, '').trim(); if (!/^(?:Matches your stated intent\b|Plan exits now\b|Exit the plan now\b)/i.test(description) || !/(?:^|[.!]\s+)(?:Plan exits now|Exit the plan now) with all DX (?:decisions and tasks|tasks and decisions) recorded[;.]\s*(?:nothing else is started|no further review is started)\./i.test(description)) return false; // A completed recap cannot conceal another plan decision or an instruction // to change the plan before the selected manual exit. Quoted archive lines // do not establish current obligations; direct current-status quotes do. const current = `${text}\n${description}`.replace(/\byou will run subsequent reviews yourself\b/gi, 'manual follow-up'); if (/(?:^|\n|[.!;]\s+)(?:Source|Example|Historical(?: review)?|Previously|Earlier review(?: assessment)?):/i.test(current)) return false; if (/\b(?:This|The) (?:manual |DX )?handoff (?:is|has been) [\"'`]?(?:cancell?ed|withdrawn|retracted|superseded|not current)\b/i.test(current)) return false; return !/\b(?:unresolved|outstanding)\b|\b(?:not|never)\s+(?:all\s+)?(?:done|complete|completed|recorded|written|resolved)\b|\b(?:review|findings?|decisions?|tasks?|plan)\b[^.!?\n]{0,70}\b(?:pending|remaining|withdrawn|retracted|superseded)\b|\b(?:only|complete)\s+(?:after|if|when|once)\b/i.test(current) && !/(?:^|[.!?;]\s+|\n|\b(?:should|must|need to|will)\s+)(?:(?:we|you|please|first|then|also)\s+)*(?:add|fix|edit|update|rewrite|remove|implement|resolve|decide|change|approve|start|run)\b/im.test(current); } /** The next gate may validate architecture already decided by this DX review. */ function architecturalDxTaskNavigation(call: NativePlanQuestionCall): boolean { const q = call.questions[0]!; if (call.answered !== true || call.failed !== false || !call.sessionId || !call.toolUseId || q.multiSelect || q.header.trim() !== 'Next steps' || q.options.length !== 3 || new Set(q.options.map(o => o.label)).size !== 3 || Object.keys(call.answers ?? {}).length !== 1) return false; const compact = (s: string | undefined) => (s ?? '').replace(/\s+/g, ' ').trim(); const match = /^D[1-9]\d* [—–-] Next steps: DX Review is complete \((?:10|[0-9])\/10 → (?:10|[0-9])\/10, ([1-9]\d*) P1 tasks, TTHW target achievable\)\. The ([1-9]\d*) fixes include architectural decisions \(demo CI exemption, arg order normalization\) that should go through an engineering gate\. What next\? $/.exec(compact(q.question)); if (!match || match[1] !== match[2]) return false; const labels = q.options.map(o => o.label.trim().replace(/\s*\(Recommended\)$/, '')); const expected = ['Run /plan-eng-review next', 'Ready to implement — run /devex-review after shipping', "Skip, I'll handle next steps manually"]; const descriptions = [ 'The demo CI exemption and argument order change are architectural decisions. Eng review validates the approach before implementation and is the required shipping gate.', `Skip eng review and implement the ${match[1]} tasks directly. Run /devex-review on the live SDK to verify the TTHW target was actually hit.`, 'Take the plan file and implementation tasks and proceed independently.', ]; // Consume every offered description in its own navigation role; an appended // remedy or unapproved scope change still requires a later report write. return new Set(labels).size === 3 && labels.every((label, index) => { const role = expected.indexOf(label); return role >= 0 && compact(q.options[index]!.description) === descriptions[role]; }) && q.options.some(o => o.label === call.answers?.[q.question]); } /** A completed DX recap can offer navigation over already specified tasks. */ function specifiedDxTaskNavigation(call: NativePlanQuestionCall): boolean { const q = call.questions[0]!; if (call.answered !== true || call.failed !== false || !call.sessionId || !call.toolUseId || q.multiSelect || q.header.trim() !== 'Next steps' || q.options.length !== 3 || new Set(q.options.map(o => o.label)).size !== 3 || Object.keys(call.answers ?? {}).length !== 1) return false; const compact = (value: string | undefined) => (value ?? '').replace(/\s+/g, ' ').trim(); const question = compact(q.question); const match = /^D\d+ [—–-] What's next\? DX review complete\. ([1-9]\d*) tasks specified \(([1-9]\d*) P1 block ship, ([1-9]\d*) P2 same branch\)\. DX score: (?:10|[0-9])\/10 → (?:10|[0-9])\/10\. TTHW: \d+(?:\.\d+)? min → < (\d+(?:\.\d+)?) min \(Champion tier\) after T\d+ lands\. The API changes \((T\d+, T\d+, T\d+)\) have architectural implications that warrant an eng review\. $/.exec(question); if (!match || Number(match[1]) !== Number(match[2]) + Number(match[3]) || Number(match[4]) <= 0) return false; const labels = q.options.map(o => o.label.trim().replace(/\s*\(Recommended\)$/, '')); const expected = ['Run /plan-eng-review next', 'Ready to implement', 'Skip — handle next steps manually']; const descriptions = [ `API changes (${match[5]}) have architectural implications. Eng review validates the --skip-ci-check design, deprecation shim contract, and argument-order change scope before implementation.`, `Skip eng review. Start implementing T1–T${match[1]} directly. Run /devex-review after shipping to measure the real TTHW against the < ${match[4]} min target.`, 'No follow-up review needed right now.', ]; return labels.every((label, index) => { const role = expected.indexOf(label); return role >= 0 && compact(q.options[index]!.description) === descriptions[role]; }) && new Set(labels).size === 3 && q.options.some(o => o.label === call.answers?.[q.question]); } /** Completed issue decisions can hand off their existing numbered tasks. */ function resolvedDxTaskNavigation(call: NativePlanQuestionCall): boolean { const q = call.questions[0]!; if (call.failed !== false || q.multiSelect || !/^Next steps$/i.test(q.header.trim()) || q.options.length !== 4 || new Set(q.options.map(option => option.label)).size !== 4) return false; const question = q.question.replace(/^D\s*\d+\s*[—–:-]\s*/i, '').replace(/\s+/g, ' ').trim(); const count = String.raw`(?:[1-9]\d*|one|two|three|four|five|six|seven|eight|nine|ten)`; const match = new RegExp(String.raw`^Next steps: (${count}) P1 DX tasks are ready to implement The DX review found and resolved (${count}) P1 issues\. All decisions were made \(D(\d+)[–-]D(\d+)\)\. The implementation tasks \(T(\d+)[–-]T(\d+)\) are waiting\. The plan also needs an Eng Review before shipping\. What would you like to do next\? $`, 'i').exec(question); if (!match) return false; const number = (value: string) => /^\d+$/.test(value) ? Number(value) : ['one','two','three','four','five','six','seven','eight','nine','ten'].indexOf(value.toLowerCase()) + 1; const n = number(match[1]!); if (number(match[2]!) !== n || Number(match[4]) - Number(match[3]) + 1 !== n || Number(match[6]) - Number(match[5]) + 1 !== n) return false; const labels = q.options.map(option => option.label.trim().replace(/\s*\(recommended\)\s*$/i, '')); const expected = ['Run /plan-eng-review next', `Start implementing T${match[5]}–T${match[6]} now`, 'Run /devex-review after shipping', "Done for now — I'll handle next steps manually"]; const descriptions = [ /^DX fixes touch [\w./-]+(?: \([\w -]+(?:, [\w -]+)*\))?(?: and [\w./-]+)*\. Eng review validates the implementation approach for those changes\.$/i, /^The tasks are well-defined\. Jump straight to implementation and run \/plan-eng-review after\.$/i, /^Implement the tasks and then run \/devex-review on the live SDK to verify TTHW actually hits the <\d+(?:\.\d+)?-minute target\.$/i, /^Save the plan and review report; return to it when ready\.$/i, ]; return labels.every((label, index) => { const kind = expected.findIndex(expected => expected.toLowerCase() === label.replace(/T(\d+)-T(\d+)/g, 'T$1–T$2').toLowerCase()); return kind >= 0 && descriptions[kind]!.test(q.options[index]!.description?.trim() ?? ''); }) && q.options.some(option => option.label === call.answers?.[q.question]); } /** * Test helper: if `obs.planFile` was set, read it and assert * `## GSTACK REVIEW REPORT` is the last `## ` section. Throws on * violation with a diagnostic message including the plan path, * the reason, any trailing headings, and the last 2KB of TTY output. * * Used by the four plan-mode E2E tests * (skill-e2e-plan-{eng,ceo,design,devex}-plan-mode.test.ts) to enforce * the {{PLAN_FILE_REVIEW_REPORT}} resolver contract uniformly. Gates on * `obs.planFile` (artifact existing), not on `obs.outcome === 'plan_ready'`, * so it also catches the report-missing case under `'asked'` / * `'wrote_findings_before_asking'` when a plan was already written. */ export function assertReportAtBottomIfPlanWritten( obs: { planFile?: string; evidence: string; outcome?: string }, ): void { if (!obs.planFile) return; // Skip when the plan file path was detected from TTY output but no file // exists on disk. This happens when the model mentions a path mid-stream // (e.g., as a tool-call argument that was interrupted, or in a draft that // was never persisted). The report-at-bottom contract is for fully-written // plan files; ENOENT means there's no file content to enforce against. if (!fs.existsSync(obs.planFile)) return; // Skip on 'asked' outcomes — these are smoke tests that exited at the // first AUQ render (Step 0 only). The model never reached the workflow's // report-writing step, so a partial plan file without the report section // is the expected mid-flight state, not a contract violation. The // report-at-bottom check applies to outcomes that imply the workflow // ran end-to-end (plan_ready, completion_summary, etc.). if (obs.outcome === 'asked') return; const content = fs.readFileSync(obs.planFile, 'utf-8'); const verdict = assertReviewReportAtBottom(content); if (!verdict.ok) { const trailing = verdict.trailingHeadings?.length ? `\ntrailing headings: ${verdict.trailingHeadings.join(', ')}` : ''; throw new Error( `GSTACK REVIEW REPORT contract violation in ${obs.planFile}: ${verdict.reason}${trailing}\n` + `--- evidence (last 2KB) ---\n${obs.evidence}`, ); } }