Merge capy/audit-fix-wave (#2994) into the harness branch

#2994 deletes the plan-*-finding-count evals, ceo-payment-findings.ts and
design-count-review.ts. Drop the CEO throw diagnostics and Design boundary
work with them, and drop the structured completion predicate, stopReason,
review-log binding and plan/review-log evidence copy: no surviving
runPlanSkillCounting caller passes expectedPlanPath, so they would be dead
code. Keep idleFor in timeout summaries (every counting caller can time
out), asserted in the existing timeout test. W7 and W8 are unchanged.
This commit is contained in:
garrytan committed 2026-09-29 15:14:06 +00:00
commit 2e1dff825d
739 files changed
+31212 -108184

No files matched your search

-6
View File
@@ -402,12 +402,6 @@ export function verboseSkill(): string {
);
}
function execGit(args: string[]): string {
const r = spawnSync('git', args, { cwd: ROOT, encoding: 'utf-8', maxBuffer: 64 * 1024 * 1024, timeout: 30_000 });
if (r.status !== 0) throw new Error(`git ${args.join(' ')} failed: ${r.stderr}`);
return r.stdout;
}
/**
* Capture the real mode-selection tool input, without answering the question.
* Print mode retains the existing 12-turn cap; interactive CLI max-turns is
-88
View File
@@ -97,91 +97,3 @@ export function createAutoplanEditDigest(file: string, removed: string, added: s
})};
return digest;
}
/** Require complete numbered rows and the marker column of this exact diff. */
export function matchesAutoplanDigestRows(rows: string[], before: Buffer, digest: AutoplanEditDigest,
allowReplacementReset = false): boolean {
if (!validAutoplanEditDigest(digest) || sha(before) !== digest.beforeSHA256) return false;
const fullRow = /^( {0,3})([1-9]\d*) ([+ -])(.*)$/;
const first = rows.findIndex(row => fullRow.test(row));
const leading = first > 0 ? rows.slice(0, first) : [];
const chunks: Array<{kind: string; text: string; line: number}> = [];
let column: number | undefined, previousLine = 0;
let removedStart: number | undefined, replacementReset = false;
for (const row of leading.length ? rows.slice(first) : rows) {
const full = fullRow.exec(row);
if (full) {
const line = Number(full[2]), markerColumn = full[1]!.length + full[2]!.length + 1;
const kind = full[3]!, previousKind = chunks.at(-1)?.kind;
// The complete native replacement panel numbers the old block first,
// then restarts additions at that block's first line. Only the caller
// that owns this full panel opts in; all row hashes still must match.
const reset = allowReplacementReset && !replacementReset && previousKind === '-' && kind === '+' && line === removedStart;
if (!Number.isSafeInteger(line) || (line < previousLine && !reset) ||
(column !== undefined && column !== markerColumn)) return false;
if (allowReplacementReset) {
if ((kind === '-' || kind === '+') && kind === previousKind && line !== previousLine + 1) return false;
if (kind === '-' && previousKind !== '-') {
if (removedStart !== undefined || chunks.some(c => c.kind === '+')) return false;
removedStart = line;
}
if (previousKind === '-' && kind !== '-' && !reset) return false;
if (kind === '+' && previousKind !== '+' && removedStart !== undefined && !reset) return false;
if (reset) replacementReset = true;
}
column = markerColumn; previousLine = line;
chunks.push({kind, text: full[4]!, line});
} else {
if (column === undefined || !row.startsWith(' '.repeat(column))) return false;
const last = chunks.at(-1), kind = row[column];
if (!last || kind !== last.kind) return false;
last.text += row.slice(column + 1);
}
}
const originals = new Set(before.toString('utf8').split('\n').map(autoplanEditLineHash));
if (allowReplacementReset) {
const oldRows = chunks.filter(c => c.kind !== '+').map(c => autoplanEditLineHash(c.text));
const newRows = chunks.filter(c => c.kind !== '-').map(c => autoplanEditLineHash(c.text));
const starts = (rows: string[], hashes: string[]) => rows.flatMap((_, index) =>
hashes.every((hash, offset) => rows[index + offset] === hash) ? [index] : []);
const oldStarts = starts(oldRows, digest.oldLineHashes), newStarts = starts(newRows, digest.newLineHashes);
if (oldStarts.length !== 1 || newStarts.length !== 1 || oldStarts[0] !== newStarts[0]) return false;
const originalLines = before.toString('utf8').split('\n').map(autoplanEditLineHash);
const delta = digest.newLineHashes.length - digest.oldLineHashes.length;
let oldLine = chunks[0]!.line, newLine = oldLine, added = false;
for (const row of chunks) {
if (row.kind !== '+') {
if (row.line !== oldLine + (added ? delta : 0) || autoplanEditLineHash(row.text) !== originalLines[oldLine - 1]) return false;
oldLine++;
}
if (row.kind !== '-' && row.line !== newLine++) return false;
if (row.kind === '+') added = true;
}
}
let authenticatedClip = false;
if (leading.length) {
const c = digest.clippedAdditions, next = chunks[0];
if (!c || c.status !== 'complete' || column === undefined || !next || chunks.length < 2 ||
leading.length > MAX_SUFFIX_SCALARS || leading.some(row =>
!row.startsWith(' '.repeat(column)) || row[column] !== '+')) return false;
const fragment = leading.map(row => row.slice(column + 1)).join('').replace(/\s/g, '');
const length = Array.from(fragment).length, record = c.lines.find(row => row.line === next.line - 1);
if (!record || !length || length > MAX_SUFFIX_SCALARS || record.nextLineHash !== autoplanEditLineHash(next.text) ||
originals.has(record.lineHash) || digest.oldLineHashes.includes(record.lineHash) ||
record.suffixHashes[length - 1] !== suffixHash(record.line, record.lineHash, record.nextLineHash, fragment)) return false;
const beforeLines = before.toString('utf8').split('\n');
if (chunks.some((row, index) => {
if (row.line !== next.line + index || row.kind === '-') return true;
const relative = row.line - c.startLine, hash = autoplanEditLineHash(row.text);
if (relative < digest.newLineHashes.length) return digest.newLineHashes[relative] !== hash;
const originalIndex = row.line - (digest.newLineHashes.length - digest.oldLineHashes.length) - 1;
return row.kind !== ' ' || originalIndex < 0 || originalIndex >= beforeLines.length ||
autoplanEditLineHash(beforeLines[originalIndex]!) !== hash;
})) return false;
authenticatedClip = true;
}
return chunks.length >= 2 && (authenticatedClip || chunks.some(c => c.kind === '+' && /\S/.test(c.text) &&
!originals.has(autoplanEditLineHash(c.text)) && !digest.oldLineHashes.includes(autoplanEditLineHash(c.text)))) &&
chunks.every(c => c.kind === '+' ? digest.newLineHashes.includes(autoplanEditLineHash(c.text)) :
originals.has(autoplanEditLineHash(c.text)) && (c.kind !== '-' || digest.oldLineHashes.includes(autoplanEditLineHash(c.text))));
}
@@ -1,8 +1,6 @@
/** One-time input for a cropped native Edit of an already-owned review artifact. */
import * as fs from 'node:fs';
import * as path from 'node:path';
import { createHash } from 'node:crypto';
import { validAutoplanEditDigest, readAutoplanDigestFile, matchesAutoplanDigestRows, createAutoplanEditDigest, autoplanEditLineHash } from './autoplan-artifact-digest';
import type { PendingAutoplanArtifact } from './autoplan-artifact-recorder';
import type { NativePublicToolEvent } from './plan-count-transcript';
@@ -19,105 +17,6 @@ interface ArtifactPermissionContext {
}
const MAX_BYTES = 1024 * 1024;
const compact = (text: string) => text.replace(/\s/g, '');
/** A completed write distinguishes a new same-looking file confirmation. */
export function autoplanPermissionProgressKey(viewport: string, events: readonly NativePublicToolEvent[]): string | undefined {
const file = /^ {0,3}Do you want to (?:create|overwrite|edit) ([^\n?]+)\? *$/m.exec(viewport)?.[1];
if (!file || new Set(events.map(event => event.sessionId)).size !== 1) return;
const menu = compact(viewport);
for (let i = events.length - 1; i >= 0; i--) {
const result = events[i]!;
if (result.kind !== 'result' || result.isError !== false) continue;
const uses = events.slice(0, i).filter(event => event.kind === 'use' &&
event.sessionId === result.sessionId && event.toolUseId === result.toolUseId);
if (uses.length !== 1) continue;
const use = uses[0]!, target = use.input?.file_path;
if (!['Write', 'Edit'].includes(use.name ?? '') || typeof target !== 'string' ||
!path.isAbsolute(target) || path.basename(target) !== file ||
!menu.includes(`alwaysallowaccessto${compact(path.dirname(target))}forthissession`) ||
!Number.isFinite(Date.parse(use.timestamp)) || Date.parse(result.timestamp) < Date.parse(use.timestamp) ||
!Number.isFinite(Date.parse(result.timestamp))) continue;
return `${result.sessionId}:${result.toolUseId}`;
}
}
/** The native header may remain above the diff; both displayed paths must bind. */
function ownedEditDiffRows(rows: string[], file: string, ownedStateRoot?: string): string[] | null {
const header = rows.findIndex(row => /^[●⏺] Update\(/.test(row));
if (header > 0 && rows.slice(0,header).some(row => row.trim())) {
// A completed native tool's diff may remain above the active edit panel.
// Only its indented diff output is ignored; competing panels or prose are
// not evidence for the current request and cannot be used as a prefix.
const prefix = rows.slice(0, header), fullRow = /^ {6}([1-9]\d*) ([+ -])/;
const first = prefix.map(row => fullRow.exec(row)).find(Boolean);
if (!first) return null;
const markerColumn = 6 + first[1]!.length + 1;
let kind: string | undefined;
for (const row of prefix) {
if (!row.trim()) continue;
const full = fullRow.exec(row);
if (full) {
if (!Number.isSafeInteger(Number(full[1])) || 6 + full[1]!.length + 1 !== markerColumn) return null;
kind = full[2];
} else {
const wrappedKind = row[markerColumn];
if (!row.startsWith(' '.repeat(markerColumn)) || !['+', ' ', '-'].includes(wrappedKind ?? '') ||
(kind !== undefined && wrappedKind !== kind)) return null;
kind = wrappedKind;
}
}
rows = rows.slice(header);
}
// A redraw can repeat the same native tool title above one current panel.
// Those homogeneous titles supply no authority: the full panel below must
// still bind its path, current request, content, and exact one-time menu.
const repeated: string[] = [];
let panelAt = 0;
for (; panelAt < rows.length; panelAt++) {
if (!rows[panelAt]!.trim()) continue;
const title = /^[●⏺] Update\(([^\n]+)\)$/.exec(rows[panelAt]!);
if (!title) break;
repeated.push(title[1]!);
}
if (repeated.length > 1 && ownedStateRoot && /^[─╌]{8,}$/.test(rows[panelAt] ?? '') &&
rows[panelAt + 1]?.trim() === 'Edit file') {
const relative = path.relative(ownedStateRoot, file).split(path.sep).join('/');
const alias = path.basename(ownedStateRoot) === '.gstack' ? `~/.gstack/${relative}` : undefined;
if (repeated.some(title => title !== repeated[0]) || (repeated[0] !== file && repeated[0] !== alias)) return null;
rows = rows.slice(panelAt);
}
if (rows.filter(row => /^[●⏺] Update\(/.test(row)).length > 1) return null;
// A viewport can start at the native Edit panel after its tool title has
// scrolled away. The remaining displayed path must still bind the complete
// owned project/artifact path; the menu and current request are checked below.
if (/^[─╌]{8,}$/.test(rows[0] ?? '') && rows[1]?.trim() === 'Edit file') {
if (!ownedStateRoot || !/^[─╌]{8,}$/.test(rows[3] ?? '')) return null;
const relative = path.relative(ownedStateRoot, file).split(path.sep).join('/');
const alias = path.basename(ownedStateRoot) === '.gstack' ? `~/.gstack/${relative}` : undefined;
const displayed = rows[2]?.trim() ?? '';
if (displayed !== file && displayed !== alias) {
const suffix = displayed.startsWith('…') ? displayed.slice(1).split(path.sep).join('/') : '';
if ((suffix !== relative && !suffix.endsWith('/' + relative)) || !file.split(path.sep).join('/').endsWith(suffix)) return null;
}
return rows.slice(4);
}
const update = /^[●⏺] Update\(([^\n]+)\)$/.exec(rows[0] ?? '');
if (!update) return rows; // Existing cropped-only row guards still apply.
if (!ownedStateRoot || rows[1]?.trim() !== '' || !/^[─╌]{8,}$/.test(rows[2] ?? '') ||
rows[3]?.trim() !== 'Edit file' || !/^[─╌]{8,}$/.test(rows[5] ?? '')) return null;
const relative = path.relative(ownedStateRoot,file).split(path.sep).join('/');
const alias = path.basename(ownedStateRoot) === '.gstack' ? `~/.gstack/${relative}` : undefined;
if (update[1] !== file && update[1] !== alias) return null;
const displayed = rows[4]?.trim() ?? '';
if (displayed !== file && displayed !== alias) {
const suffix = displayed.startsWith('…') ? displayed.slice(1).split(path.sep).join('/') : '';
// A truncated prefix must still retain the complete owned project/artifact
// path. A basename or sibling-project suffix cannot bind this request.
if ((suffix !== relative && !suffix.endsWith('/'+relative)) || !file.split(path.sep).join('/').endsWith(suffix)) return null;
}
return rows.slice(6);
}
export function ownedAutoplanArtifact(file: string, context: Pick<ArtifactPermissionContext, 'cwd' | 'ownedStateRoot'>): boolean {
if (!context.ownedStateRoot || !path.isAbsolute(file) || path.resolve(file) !== file) return false;
@@ -137,281 +36,6 @@ export function ownedAutoplanArtifact(file: string, context: Pick<ArtifactPermis
} catch { return false; }
}
/** Require the whole current cropped diff, exact menu, and requested edit text. */
function matchesCroppedEdit(viewport: string, file: string, before: string, removed: string, after: string,
ownedStateRoot?: string): boolean {
if (viewport.length > MAX_BYTES) return false;
const text = viewport.replace(/\r\n?/g, '\n');
const menu = /^ {0,3}Do you want to make this edit to ([^\n?]+)\? *\n {0,3}❯ *1\. Yes *\n {0,3}2\. Yes, and switch to accept edits \(auto-approve file edits and common file commands\) for this session(?: \(shift\+tab\))? *\n {0,3}3\. No *\n\s*Esc to cancel [·•] Tab to amend\s*$/m.exec(text);
if (!menu || menu.index + menu[0].length !== text.length || menu[1] !== path.basename(file)) return false;
const rows = text.slice(0, menu.index).trimEnd().split('\n');
if (!/^[╌─]{8,}$/.test(rows.pop() ?? '')) return false;
const diffRows = ownedEditDiffRows(rows,file,ownedStateRoot);
if (!diffRows) return false;
const chunks: Array<{ kind: string; text: string }> = [];
let markerColumn: number | undefined;
for (const row of diffRows) {
const numbered = /^( {0,3})([1-9]\d*) ([+ -])(.*)$/.exec(row);
if (numbered) {
const line = Number(numbered[2]), column = numbered[1]!.length + numbered[2]!.length + 1;
// Match the digest parser: every row owns one marker column, regardless
// of line-number width; continuations retain that column and diff kind.
if (!Number.isSafeInteger(line) || (markerColumn !== undefined && markerColumn !== column)) return false;
markerColumn = column;
chunks.push({ kind: numbered[3]!, text: numbered[4]! });
} else {
const last = chunks.at(-1);
if (markerColumn === undefined || !row.startsWith(' '.repeat(markerColumn)) ||
!last || row[markerColumn] !== last.kind) return false;
last.text += row.slice(markerColumn + 1);
}
}
// The crop itself cannot contain an example introduction, quote, unrelated
// prompt, or arbitrary diff: each row must occur in this exact pending edit.
const originals = before.split('\n').map(compact);
const deletions = removed.split('\n').map(compact);
const replacements = after.split('\n').map(compact);
const changed = chunks.some(chunk => chunk.kind !== ' ' && compact(chunk.text));
const matches = (oldRows: string[], newRows: string[]) => chunks.every(chunk =>
(chunk.kind === '+' ? newRows : chunk.kind === '-' ? oldRows : originals).includes(compact(chunk.text)));
if (changed && matches(deletions, replacements)) return true;
// Edit arguments can start or end inside a line while the native preview
// displays the whole line. Reconstruct only those unchanged edge bytes
// from the unique current old substring; no viewport text supplies them.
const at = before.indexOf(removed), end = at + removed.length;
if (!changed || !removed || at < 0 || at !== before.lastIndexOf(removed) || after.length > MAX_BYTES) return false;
const prefix = before.slice(before.slice(0, at).lastIndexOf('\n') + 1, at);
const newline = before.indexOf('\n', end);
const suffix = before.slice(end, newline < 0 ? before.length : newline);
if (!prefix && !suffix) return false;
const oldRows = (prefix + removed + suffix).split('\n').map(compact);
const newRows = (prefix + after + suffix).split('\n').map(compact);
return matches(oldRows, newRows) && chunks.some(chunk =>
chunk.kind === '+' ? !oldRows.includes(compact(chunk.text)) :
chunk.kind === '-' && !newRows.includes(compact(chunk.text)));
}
export function autoplanArtifactPermissionInput(
viewport: string, context: ArtifactPermissionContext, seen: ReadonlySet<string>,
): { input: '1\r'; signature: string; file: string } | null {
const now = context.now ?? Date.now();
if (context.transcriptStatus !== 'ready' || !Number.isFinite(context.commandStartedAt) ||
context.commandStartedAt > now || context.publicTools.length > 10_000 ||
context.publicTools.some(event => !Number.isFinite(Date.parse(event.timestamp)))) return null;
const events = context.publicTools.filter(event => Date.parse(event.timestamp) >= context.commandStartedAt);
if (!events.length || events.some(event => !event.sessionId || !event.toolUseId ||
!Number.isFinite(Date.parse(event.timestamp)) || Date.parse(event.timestamp) > now) ||
new Set(events.map(event => event.sessionId)).size !== 1) return null;
// Bind the latest file mutation, which must be the sole unresolved Write/Edit.
// Claude may publish a queued Bash while its current Edit permission is open;
// that unrelated request supplies no file permission authority.
const edit = events.filter(event => event.kind === 'use' &&
(event.name === 'Write' || event.name === 'Edit')).at(-1);
if (!edit || edit.kind !== 'use' || edit.name !== 'Edit' || typeof edit.input?.file_path !== 'string' ||
typeof edit.input.old_string !== 'string' || !edit.input.old_string ||
typeof edit.input.new_string !== 'string' || edit.input.new_string === edit.input.old_string ||
(edit.input.replace_all !== undefined && edit.input.replace_all !== false)) return null;
const signature = `${edit.sessionId}:${edit.toolUseId}`;
if (seen.has(signature) || !ownedAutoplanArtifact(edit.input.file_path, context)) return null;
const uses = new Map<string, NativePublicToolEvent>();
const results = new Map<string, NativePublicToolEvent>();
let previousTime = context.commandStartedAt;
for (const event of events) {
const time = Date.parse(event.timestamp);
if (time < previousTime) return null;
previousTime = time;
const map = event.kind === 'use' ? uses : results;
if (map.has(event.toolUseId)) return null;
map.set(event.toolUseId, event);
if (event.kind === 'result' && !uses.has(event.toolUseId)) return null;
}
const writes = [...uses.values()].filter(event => event.name === 'Write' || event.name === 'Edit');
if (results.has(edit.toolUseId) || writes.filter(event => !results.has(event.toolUseId)).length !== 1) return null;
if (!writes.some(event => event.toolUseId !== edit.toolUseId &&
event.input?.file_path === edit.input!.file_path && results.has(event.toolUseId) &&
results.get(event.toolUseId)!.isError === false)) return null;
try {
const before = fs.readFileSync(edit.input.file_path, 'utf8');
if (!before.includes(edit.input.old_string) ||
!matchesCroppedEdit(viewport, edit.input.file_path, before, edit.input.old_string, edit.input.new_string,
context.ownedStateRoot)) return null;
return { input: '1\r', signature, file: edit.input.file_path };
} catch { return null; }
}
/** A previously granted viewport cannot establish a newer unpublished request. */
export const autoplanArtifactMenuKey = (viewport: string) =>
`menu:${createHash('sha256').update(viewport.replace(/\r\n?/g, '\n')).digest('hex')}`;
/** Queued native-plan edits remain unstarted; this never grants their input. */
function ownedQueuedNativePlan(file: unknown, root: string | undefined, pendingTime: number): file is string {
if (!root || typeof file !== 'string' || !path.isAbsolute(root) || path.resolve(root) !== root ||
!path.isAbsolute(file) || path.resolve(file) !== file || path.dirname(file) !== root ||
!/^[a-z][a-z0-9-]*\.md$/.test(path.basename(file))) return false;
try {
return fs.lstatSync(root).isDirectory() && fs.lstatSync(file).isFile() &&
fs.realpathSync(file) === path.join(fs.realpathSync(root), path.basename(file)) && fs.statSync(file).size <= MAX_BYTES &&
Math.floor(fs.statSync(file).mtimeMs) <= pendingTime;
} catch { return false; }
}
/** Bind completed snapshot output and queued native-plan redraw labels before
* the full current panel. Arbitrary prose, examples and competing panels stay. */
function queuedPlanViewport(viewport: string, queuedPlans: number, current: NativePublicToolEvent,
events: NativePublicToolEvent[]): string {
if (!queuedPlans) return viewport;
const lines = viewport.replace(/\r\n?/g, '\n').split('\n');
const title = lines.findIndex(line => /^[●⏺] Update\(/.test(line));
if (title < 0 || lines.filter(line => /^[●⏺] Update\(/.test(line)).length !== 1) return viewport;
const panel = lines.findIndex((line, i) => i > title && /^[─╌]{8,}$/.test(line));
const redraws = lines.slice(title + 1, panel).filter(line => line.trim());
if (panel < 0 || redraws.length !== queuedPlans || redraws.some(line => !/^[●⏺] Updated plan$/.test(line))) return viewport;
const prefix = lines.slice(0, title);
while (prefix.at(-1)?.trim() === '') prefix.pop();
if (prefix.some(line => line.trim())) {
const prior = events.slice(0, events.indexOf(current));
const result = prior.filter(event => event.kind === 'result').at(-1);
const use = result && prior.find(event => event.kind === 'use' && event.toolUseId === result.toolUseId);
const command = /^ {8}"([^"\n]+)…\)$/.exec(prefix[0] ?? '');
const outputEnd = prefix.length - 2;
if (!command || !use || use.name !== 'Bash' || use.messageId !== current.messageId ||
use.requestId !== current.requestId || typeof use.input?.command !== 'string' ||
!use.input.command.includes(command[1]!) || result?.isError !== false || typeof result.content !== 'string' ||
!/^ {2}⎿[ \u00a0]+\{$/.test(prefix[1] ?? '') ||
!/^ {5}… \+[1-9]\d* lines \(ctrl\+o to expand\)$/.test(prefix[outputEnd] ?? '') ||
!/^ {2}⎿[ \u00a0]+Allowed by auto mode classifier$/.test(prefix.at(-1) ?? '') || outputEnd < 3) return viewport;
const displayed = ['{', ...prefix.slice(2, outputEnd).map(line => /^ {5}( {2}\S.*)$/.exec(line)?.[1])];
if (displayed.some(line => line === undefined) ||
result.content.split('\n').slice(0, displayed.length).join('\n') !== displayed.join('\n')) return viewport;
}
return [lines[title], '', ...lines.slice(panel)].join('\n');
}
/** Native batch redraws are display-only: bind their titles, waiting command,
* and one clipped context row to public events before removing the prefix. */
function queuedArtifactViewport(viewport: string, current: NativePublicToolEvent,
events: NativePublicToolEvent[], queued: ReadonlySet<string>, file: string,
pendingTime: number, ownedStateRoot?: string): string {
if (!queued.size || !ownedStateRoot) return viewport;
const successors = events.filter(e => e.kind === 'use' && queued.has(e.toolUseId));
if (successors.some(e => e.input?.file_path !== file)) return viewport;
const lines = viewport.replace(/\r\n?/g, '\n').split('\n');
const firstTitle = lines.findIndex(line => /^[●⏺] Update\(/.test(line));
const panel = lines.findIndex((line, i) => i > firstTitle && /^[─╌]{8,}$/.test(line));
if (firstTitle < 1 || panel < 0) return viewport;
const prefix = lines.slice(0, firstTitle).filter(line => line.trim());
const clipped = prefix.length === 1 && /^ {10}(\S.{15,})$/.exec(prefix[0]!);
if (!clipped) return viewport;
const relative = path.relative(ownedStateRoot, file).split(path.sep).join('/');
const alias = path.basename(ownedStateRoot) === '.gstack' ? `~/.gstack/${relative}` : undefined;
const rows = lines.slice(firstTitle, panel).filter(line => line.trim());
const titles = rows.slice(0, successors.length + 1);
if (titles.length !== successors.length + 1 || titles.some(row => {
const title = /^[●⏺] Update\(([^\n]+)\)$/.exec(row);
return !title || (title[1] !== file && title[1] !== alias);
})) return viewport;
const bash = rows.slice(titles.length);
if (bash.length < 2 || !/^ {2}⎿[ \u00a0]+Waiting…$/.test(bash.at(-1)!)) return viewport;
const parts = bash.slice(0, -1).map((row, i) =>
(i === 0 ? /^[●⏺] Bash\((.+)$/ : /^ {6}(.+)$/).exec(row)?.[1]);
if (parts.some(part => part === undefined)) return viewport;
const rendered = parts.join('');
if (!rendered.endsWith('…)')) return viewport;
const commandPrefix = compact(rendered.slice(0, -2));
const waiting = events.filter(e => e.kind === 'use' && e.name === 'Bash' &&
e.messageId === current.messageId && e.requestId === current.requestId &&
events.indexOf(e) > Math.max(...successors.map(s => events.indexOf(s))) &&
!events.some(result => result.kind === 'result' && result.toolUseId === e.toolUseId) &&
typeof e.input?.command === 'string' && compact(e.input.command).startsWith(commandPrefix));
if (commandPrefix.length < 32 || waiting.length !== 1) return viewport;
const completed = events.filter(e => e.kind === 'result' && e.isError === false &&
Date.parse(e.timestamp) <= pendingTime).map(result => ({result, use:events.find(e =>
e.kind === 'use' && e.toolUseId === result.toolUseId)})).filter(({use}) =>
use?.name === 'Edit' && use.input?.file_path === file &&
use.messageId === current.messageId && use.requestId === current.requestId).at(-1);
const replacement = completed?.use?.input?.new_string;
if (typeof replacement !== 'string' || !replacement) return viewport;
const before = fs.readFileSync(file, 'utf8'), at = before.indexOf(replacement);
if (at < 0 || before.indexOf(replacement, at + 1) !== -1) return viewport;
// Native diffs display at most three unchanged context lines after an edit.
// The cropped row must be a suffix of one of those current, unchanged lines.
const lineEnd = before.indexOf('\n', at + replacement.length);
const context = lineEnd < 0 ? [] : before.slice(lineEnd + 1).split('\n').slice(0, 3);
if (!context.some(line => compact(line).endsWith(compact(clipped[1]!)))) return viewport;
return lines.slice(panel).join('\n');
}
/** A cropped command caption is display only. Bind the complete wrapped command
* to one unstarted successor in the current published batch before discarding it. */
function queuedCommandViewport(viewport: string, current: NativePublicToolEvent,
events: NativePublicToolEvent[], queued: ReadonlySet<string>, hookSeenIds: readonly string[],
viewportCapturedAt: number): string {
if (!queued.size) return viewport;
const text = viewport.replace(/\r\n?/g, '\n');
const panels = [...text.matchAll(/^[─╌]{8,}\n {0,3}Edit file[ \t]*\n/gm)];
if (panels.length !== 1 || panels[0]!.index === 0) return viewport;
const rows = text.slice(0, panels[0]!.index).split('\n');
while (rows.at(-1)?.trim() === '') rows.pop();
const parts = rows.map((row, i) =>
(i === 0 ? /^ {2}⎿[ \u00a0]+\$ (\S.*)$/ : /^ {5}(\S.*)$/).exec(row)?.[1]);
if (!parts.length || parts.some(part => part === undefined)) return viewport;
const currentIndex = events.indexOf(current);
const waiting = events.filter(e => e.kind === 'use' && e.name === 'Bash' &&
e.messageId === current.messageId && e.requestId === current.requestId && events.indexOf(e) > currentIndex &&
!events.some(result => result.kind === 'result' && result.toolUseId === e.toolUseId));
const command = waiting[0], input = command?.input?.command;
if (waiting.length !== 1 || !command || typeof input !== 'string' || !input || input.length > MAX_BYTES ||
/[\x00-\x1f\x7f]/.test(input) || hookSeenIds.includes(command.toolUseId) ||
Date.parse(command.timestamp) > viewportCapturedAt ||
events.some(e => e.kind === 'use' && queued.has(e.toolUseId) && events.indexOf(e) >= events.indexOf(command))) return viewport;
// Preserve every displayed character, including spaces inside quoted arguments.
// Only whitespace omitted at a renderer soft-wrap boundary may be skipped.
let remaining = input;
for (let i = 0; i < parts.length; i++) {
if (!remaining.startsWith(parts[i]!)) return viewport;
remaining = remaining.slice(parts[i]!.length);
if (i < parts.length - 1) remaining = remaining.replace(/^[ \t]+/, '');
}
return remaining === '' ? text.slice(panels[0]!.index) : viewport;
}
/** A native command description can remain above an unpublished Edit panel.
* Its text supplies no command identity, completion, or approval authority.
* Only the digest-bound pending path may discard this one display prefix. */
function pendingCommandDisplayViewport(viewport: string, file: string, ownedStateRoot?: string): string {
const text = viewport.replace(/\r\n?/g, '\n');
const panels = [...text.matchAll(/^[─╌]{8,}\n {0,3}Edit file[ \t]*\n/gm)];
if (panels.length !== 1 || panels[0]!.index === 0) return viewport;
const prefix = text.slice(0, panels[0]!.index).split('\n').filter(line => line.trim());
// An unpublished batch can leave the current Update title, plan redraws,
// and a queued Bash card above the panel. These cards grant no authority:
// only the one current Edit's owned path and complete digest below do so.
const update = /^[●⏺] Update\(([^\n]+)\)$/.exec(prefix[0] ?? '');
if (update && ownedStateRoot) {
const relative = path.relative(ownedStateRoot, file).split(path.sep).join('/');
const alias = path.basename(ownedStateRoot) === '.gstack' ? `~/.gstack/${relative}` : undefined;
const bash = prefix.findIndex(row => /^[●⏺] Bash\(/.test(row));
const command = prefix.slice(bash, -1);
if ((update[1] === file || update[1] === alias) && bash > 1 &&
prefix.slice(1, bash).every(row => /^[●⏺] Updated plan$/.test(row)) &&
/^ {2}⎿[ \u00a0]+Waiting…$/.test(prefix.at(-1) ?? '') && command.length > 0 &&
command.every((row, i) => (i === 0 ? /^[●⏺] Bash\(\S.*$/ : /^ {6}\S.*$/).test(row)) &&
command.at(-1)!.endsWith('…)') &&
!command.slice(1).some(row => /^ {6}[●⏺❯☐□>]|^ {6}(?:`{3,}|~{3,})/.test(row)) &&
!/(?:Do you want|Would you like|Bash command[^\n]*permission|requested permissions?|allow all edits|always allow access|Esc to cancel|Edit file)/i.test(command.join('\n'))) {
return text.slice(panels[0]!.index);
}
return viewport;
}
const title = /^[●⏺] ([^\n]+)$/.exec(prefix[0] ?? '')?.[1];
if (!title || /^(?:["'`“‘]|(?:source|example|quoted|history|historical|hypothetical|previous|earlier)\b)/i.test(title) ||
!/^ {2}⎿[ \u00a0]+\$ \S.*$/.test(prefix[1] ?? '') ||
prefix.slice(2).some(line => !/^ {5}\S.*$/.test(line)) ||
prefix.slice(1).some(line => /^[ \t]*[●⏺❯☐□>]|^[ \t]*(?:`{3,}|~{3,})/.test(line)) ||
/(?:Do you want|Would you like|Bash command[^\n]*permission|requested permissions?|allow all edits|always allow access|Esc to cancel|Edit file)/i.test(prefix.join('\n'))) return viewport;
return text.slice(panels[0]!.index);
}
/** Same native-history gate for an unpublished pending Edit, independent of its display. */
export function hasPendingAutoplanArtifactHistory(
context: ArtifactPermissionContext & { pending?: PendingAutoplanArtifact },
@@ -444,151 +68,3 @@ export function hasPendingAutoplanArtifactHistory(
Date.parse(results.get(e.toolUseId)!.timestamp) <= pendingTime)) return false;
return true;
}
/** Metadata-only fallback. Added rows are display evidence, never request content. */
export function pendingAutoplanArtifactPermissionInput(viewport: string,
context: ArtifactPermissionContext & { pending?: PendingAutoplanArtifact; viewportCapturedAt: number },
seen: ReadonlySet<string>,
): { input: '1\r'; signature: string; file: string } | null {
const p = context.pending, now = context.now ?? Date.now();
if (p?.editDigest !== undefined && !validAutoplanEditDigest(p.editDigest)) return null;
if (!p || !Number.isFinite(now) || context.transcriptStatus !== 'ready' || !Number.isFinite(context.commandStartedAt) ||
!Number.isFinite(context.viewportCapturedAt) || context.viewportCapturedAt > now ||
context.commandStartedAt > context.viewportCapturedAt || viewport.length > MAX_BYTES ||
p.source !== 'pre_tool_use' || p.tool !== 'Edit' || typeof p.file !== 'string' ||
!/^[A-Za-z0-9_-]{1,160}$/.test(p.sessionId) || !/^[A-Za-z0-9_-]{1,160}$/.test(p.toolUseId) ||
!ownedAutoplanArtifact(p.file, context)) return null;
const pendingTime = Date.parse(p.timestamp), signature = `${p.sessionId}:${p.toolUseId}`;
if (!Number.isFinite(pendingTime) || pendingTime < context.commandStartedAt || pendingTime > context.viewportCapturedAt ||
seen.has(signature) || seen.has(autoplanArtifactMenuKey(viewport)) || context.publicTools.length > 10_000) return null;
if (!hasPendingAutoplanArtifactHistory(context)) return null;
try {
const currentViewport = p.editDigest ? pendingCommandDisplayViewport(viewport, p.file, context.ownedStateRoot) : viewport;
const text = currentViewport.replace(/\r\n?/g, '\n');
const menu = /^ {0,3}Do you want to make this edit to ([^\n?]+)\? *\n {0,3}❯ *1\. Yes *\n {0,3}2\. Yes, and switch to accept edits \(auto-approve file edits and common file commands\) for this session(?: \(shift\+tab\))? *\n {0,3}3\. No *\n\s*Esc to cancel [·•] Tab to amend\s*$/m.exec(text);
if (!menu || menu.index + menu[0].length !== text.length || menu[1] !== path.basename(p.file)) return null;
const rows = text.slice(0, menu.index).trimEnd().split('\n');
if (!/^[╌─]{8,}$/.test(rows.pop() ?? '')) return null;
const diffRows = ownedEditDiffRows(rows,p.file,context.ownedStateRoot);
if (!diffRows) return null;
if (Math.floor(fs.statSync(p.file).mtimeMs) > pendingTime) return null;
if (p.editDigest) {
const before = readAutoplanDigestFile(p.file);
if (!before || createHash('sha256').update(before).digest('hex') !== p.editDigest.beforeSHA256) return null;
if (matchesAutoplanDigestRows(diffRows,before,p.editDigest,currentViewport !== viewport)) return {input:'1\r', signature, file:p.file};
if (currentViewport !== viewport) return null; // The new prefix path requires the exact digest, including additions.
// Legacy deletion crops below must still honor the recorded request digest.
}
const originals = fs.readFileSync(p.file, 'utf8').split('\n').map(compact);
const chunks: Array<{kind:string; text:string; partial?:boolean}> = [];
const fullRow = /^( {0,3})([1-9]\d*) ([+ -])(.*)$/;
const first = diffRows.map(row => fullRow.exec(row)).find(Boolean);
if (!first) return null;
const markerColumn = first[1]!.length + first[2]!.length + 1;
let numbered = 0;
for (const row of diffRows) {
const full = fullRow.exec(row);
if (full) {
if (!Number.isSafeInteger(Number(full[2])) || full[1]!.length + full[2]!.length + 1 !== markerColumn) return null;
numbered++; chunks.push({kind:full[3]!, text:full[4]!});
} else {
const kind = row[markerColumn], text = row.slice(markerColumn + 1);
if (!row.startsWith(' '.repeat(markerColumn)) || !['+', ' ', '-'].includes(kind ?? '')) return null;
if (!chunks.length) chunks.push({kind:kind!, text, partial:true});
else {
const previous = chunks.at(-1)!;
if (previous.kind !== kind) return null;
previous.text += text;
}
}
}
// A leading cropped deletion/context fragment must be an actual suffix.
// Complete removed/context rows must occur in the current owned file.
if (numbered < 2 || !chunks.some(c => c.kind === '-' && compact(c.text)) ||
chunks.some(c => c.kind !== '+' && !originals.some(line => c.partial
? line.endsWith(compact(c.text)) : line === compact(c.text)))) return null;
const digest = p.editDigest;
if (digest && chunks.some(chunk => {
const hash = autoplanEditLineHash(chunk.text);
if (chunk.kind === '+') return chunk.partial || !digest.newLineHashes.includes(hash);
if (chunk.kind !== '-') return false; // Current-file context was checked above.
return chunk.partial
? !originals.some(line => line.endsWith(compact(chunk.text)) && digest.oldLineHashes.includes(autoplanEditLineHash(line)))
: !digest.oldLineHashes.includes(hash);
})) return null;
return {input:'1\r', signature, file:p.file};
} catch { return null; }
}
/** A native hook identifies the executing request within a published tool batch. */
export function publishedAutoplanArtifactPermissionInput(viewport: string,
context: ArtifactPermissionContext & { pending?: PendingAutoplanArtifact; viewportCapturedAt: number },
seen: ReadonlySet<string>,
): { input: '1\r'; signature: string; file: string } | null {
const p=context.pending, now=context.now??Date.now();
if (!p || p.source!=='pre_tool_use' || p.tool!=='Edit' || !validAutoplanEditDigest(p.editDigest) ||
context.transcriptStatus!=='ready' || !Number.isFinite(now) || !Number.isFinite(context.commandStartedAt) ||
!Number.isFinite(context.viewportCapturedAt) || context.commandStartedAt>context.viewportCapturedAt ||
context.viewportCapturedAt>now || context.publicTools.length>10_000 || viewport.length>MAX_BYTES ||
!/^[A-Za-z0-9_-]{1,160}$/.test(p.sessionId) || !/^[A-Za-z0-9_-]{1,160}$/.test(p.toolUseId) ||
!Array.isArray(p.hookSeenIds) || !p.hookSeenIds.length || p.hookSeenIds.length>128 ||
p.hookSeenIds.some(id=>typeof id!=='string'||!/^[A-Za-z0-9_-]{1,160}$/.test(id)) ||
new Set(p.hookSeenIds).size!==p.hookSeenIds.length || !p.hookSeenIds.includes(p.toolUseId) ||
seen.has(`${p.sessionId}:${p.toolUseId}`) || seen.has(autoplanArtifactMenuKey(viewport)) ||
!ownedAutoplanArtifact(p.file,context)) return null;
const pendingTime=Date.parse(p.timestamp);
if (!Number.isFinite(pendingTime) || pendingTime<context.commandStartedAt || pendingTime>context.viewportCapturedAt ||
context.publicTools.some(e=>!Number.isFinite(Date.parse(e.timestamp)))) return null;
const events=context.publicTools.filter(e=>Date.parse(e.timestamp)>=context.commandStartedAt);
const uses=new Map<string,NativePublicToolEvent>(), results=new Map<string,NativePublicToolEvent>();
let last=context.commandStartedAt;
for (const event of events) {
const time=Date.parse(event.timestamp), map=event.kind==='use'?uses:results;
if (event.sessionId!==p.sessionId || !event.toolUseId || time<last || time>now || map.has(event.toolUseId) ||
(event.kind==='result'&&!uses.has(event.toolUseId))) return null;
last=time;map.set(event.toolUseId,event);
}
const current=uses.get(p.toolUseId), input=current?.input;
if (!current || current.name!=='Edit' || results.has(p.toolUseId) || Date.parse(current.timestamp)>pendingTime ||
!/^msg_[A-Za-z0-9_-]{1,160}$/.test(current.messageId??'') || !/^req_[A-Za-z0-9_-]{1,160}$/.test(current.requestId??'') ||
input?.file_path!==p.file || typeof input.old_string!=='string' || !input.old_string ||
typeof input.new_string!=='string' || input.new_string===input.old_string ||
(input.replace_all!==undefined&&input.replace_all!==false)) return null;
const queued=new Set<string>();
let queuedPlans = 0;
for (const mutation of [...uses.values()].filter(e=>e.name==='Edit'||e.name==='Write')) {
const result=results.get(mutation.toolUseId);
if (result && (Date.parse(mutation.timestamp)>pendingTime || Date.parse(result.timestamp)>pendingTime)) return null;
if (mutation.toolUseId===p.toolUseId || result) continue;
// Later publications are queued only when this exact batch owns them and
// the recorder has not started them. They never supply current authority.
const nativePlan = mutation.input?.file_path !== p.file &&
ownedQueuedNativePlan(mutation.input?.file_path, context.ownedNativePlansRoot, pendingTime);
if (mutation.name!=='Edit' || (mutation.input?.file_path!==p.file && !nativePlan) ||
typeof mutation.input.old_string!=='string' || !mutation.input.old_string ||
typeof mutation.input.new_string!=='string' || mutation.input.old_string===mutation.input.new_string ||
(mutation.input.replace_all!==undefined && mutation.input.replace_all!==false) ||
mutation.messageId!==current.messageId || mutation.requestId!==current.requestId ||
Date.parse(mutation.timestamp)>context.viewportCapturedAt ||
events.indexOf(mutation)<=events.indexOf(current) || p.hookSeenIds.includes(mutation.toolUseId)) return null;
if (nativePlan) {
if (![...uses.values()].some(previous => (previous.name === 'Write' || previous.name === 'Edit') &&
previous.input?.file_path === mutation.input!.file_path &&
results.get(previous.toolUseId)?.isError === false &&
Date.parse(results.get(previous.toolUseId)!.timestamp) <= pendingTime)) return null;
queuedPlans++;
}
queued.add(mutation.toolUseId);
}
try {
if (Math.floor(fs.statSync(p.file).mtimeMs)>pendingTime) return null;
const actual=createAutoplanEditDigest(p.file,input.old_string,input.new_string), expected=p.editDigest!;
if (!actual || actual.beforeSHA256!==expected.beforeSHA256 || actual.requestSHA256!==expected.requestSHA256 ||
JSON.stringify(actual.oldLineHashes)!==JSON.stringify(expected.oldLineHashes) ||
JSON.stringify(actual.newLineHashes)!==JSON.stringify(expected.newLineHashes)) return null;
const commandViewport = queuedCommandViewport(viewport,current,events,queued,p.hookSeenIds,context.viewportCapturedAt);
const rendered = queuedArtifactViewport(commandViewport,current,events,queued,p.file,pendingTime,context.ownedStateRoot);
return autoplanArtifactPermissionInput(queuedPlanViewport(rendered,queuedPlans,current,events),{...context,
publicTools:events.filter(e=>!queued.has(e.toolUseId))},seen);
} catch { return null; }
}
-380
View File
@@ -1,380 +0,0 @@
import { readOwnedClaudeTranscript } from './owned-claude-transcript';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { createHash } from 'node:crypto';
import { isDeepStrictEqual } from 'node:util';
import { currentFilePermissionTarget, nativePermissionKey, reserveNativePermissionGrant, type readPlanSkillQuestions, type NativeFilePermissionRequest, type NativePermissionGrant } from './plan-skill-questions';
import type { ClaudePtySession } from './claude-pty-runner';
import { readQuestionEvents, readQuestionCompletionEvents, readBashEvents, readBashCompletionEvents, readBashPermissionRequestEvents, type QuestionEventSource,
type QuestionEventCall, type QuestionCompletionEventCall, type BashEventCall, type BashCompletionEventCall, type BashPermissionRequestEventCall } from './plan-skill-question-events';
/** The chain may grant file edits in its fixture and native plan directory.
* The shared reservation still requires the exact owned request and menu;
* a permission-looking screen alone never authorizes an input. */
export function reserveAutoplanFilePermission(
native: ReturnType<typeof readPlanSkillQuestions>, visible: string,
opts: { cwd: string; planDir: string; granted: Set<string>; requests: Map<string, NativePermissionGrant> },
): boolean {
if (native.pendingBytes || native.ready || native.calls.some(call => call.result === 'pending')) return false;
const pending = native.permissionRequests.filter(request => request.result === 'pending');
if (!native.permissionRequestCapture || pending.length !== 1) return false;
const request = pending[0]!;
const cwd = fs.realpathSync(opts.cwd);
if (request.cwd !== cwd) throw new Error('Autoplan file permission cwd differs from its fixture');
const file = request.input.file_path;
if (typeof file !== 'string' || !path.isAbsolute(file)) throw new Error('Autoplan file permission lacks an absolute path');
const normalized = path.normalize(file);
const root = [cwd, opts.planDir].find(root => normalized.startsWith(root + path.sep));
if (!root) throw new Error('Autoplan file permission is outside its fixture and native plan directory');
// Reject symlink escapes, including a not-yet-created file below a link.
for (let entry = normalized; entry !== path.dirname(root); entry = path.dirname(entry)) {
if (fs.lstatSync(entry, { throwIfNoEntry: false })?.isSymbolicLink()) {
throw new Error('Autoplan file permission traverses a symlink');
}
}
return reserveNativePermissionGrant(native, visible, opts.granted, opts.requests);
}
/** Recover only a clipped file identity; the existing reservation remains the
* grant authority. The three fresh paints are finite, not a promise that every
* possible diff fits. A resize is never a decision or native completion. */
export class AutoplanFilePermissionViewport {
private owner: NativeFilePermissionRequest | null = null;
private paints = 0;
inputMark = -1;
get active(): boolean { return this.owner !== null; }
constructor(private readonly opts: {
session: Pick<ClaudePtySession, 'resizeQuestionViewport' | 'mark'>;
deadlineAt: number;
granted: Set<string>;
}) {}
/** Called only after the caller's unchanged native/current-screen bracket.
* true means wait for another sample, never type a permission choice. */
async advance(native: ReturnType<typeof readPlanSkillQuestions>, frame: { text: string; rawEnd: number }): Promise<boolean> {
if (!this.owner) return false;
const owner = native.permissionRequests.find(request => request.requestId === this.owner!.requestId);
if (!owner || owner.name !== this.owner.name || owner.cwd !== this.owner.cwd
|| owner.capturedAtMs !== this.owner.capturedAtMs || !isDeepStrictEqual(owner.input, this.owner.input)
|| this.owner.nativeToolId != null && owner.nativeToolId !== this.owner.nativeToolId) {
throw new Error('Autoplan expanded file permission changed ownership or input');
}
// A hook can precede transcript persistence. Once linked, retain that ID.
if (this.owner.nativeToolId == null && owner.nativeToolId != null) this.owner.nativeToolId = owner.nativeToolId;
if (owner.result === 'error') throw new Error('Autoplan expanded file permission returned an error');
if (owner.result === 'completed') {
if (!owner.nativeToolId || !Number.isFinite(owner.nativeResultAtMs)) throw new Error('Autoplan expanded file permission lacks its successful native ACK');
const restored = await this.opts.session.resizeQuestionViewport!(120, this.opts.deadlineAt);
if (restored !== null) { this.inputMark = restored; this.owner = null; }
return true;
}
// A queued Bash cannot own this pinned file repaint. The unchanged exact
// current-card reservation below still decides the sole file grant.
if (native.permissionRequests.filter(request => request.result === 'pending').length !== 1
|| native.permissionTools.some(tool => tool.id !== owner.nativeToolId && tool.name !== 'Bash')) {
throw new Error('Ambiguous native permission owner during Autoplan viewport recovery');
}
if (native.pendingBytes || frame.rawEnd !== this.opts.session.mark() || frame.rawEnd <= this.inputMark
|| this.opts.granted.has(`request:${owner.requestId}`)) return true;
try { nativePermissionKey(owner, frame.text); return false; }
catch (error) {
if (!this.clipped(owner, frame.text) || this.paints === 3) throw error;
return this.repaint();
}
}
/** The caller first runs all existing fixture/symlink/owner/grant checks.
* A clipped identity can also fail disambiguation against a queued Bash;
* neither error authorizes input before the full file card is recovered. */
async recover(error: unknown, native: ReturnType<typeof readPlanSkillQuestions>, frame: { text: string; rawEnd: number }): Promise<boolean> {
const identityError = error instanceof Error && (error.message === 'Visible permission cannot be bound to its pending native command or file path'
|| error.message === 'Ambiguous native permission owner: multiple tools are pending'
&& native.permissionTools.some(tool => tool.name === 'Bash'));
if (!identityError || this.owner || !this.opts.session.resizeQuestionViewport || frame.rawEnd !== this.opts.session.mark()
|| native.pendingBytes || native.ready || native.calls.some(call => call.result === 'pending')) return false;
const pending = native.permissionRequests.filter(request => request.result === 'pending');
const owner = pending[0];
if (!native.permissionRequestCapture || pending.length !== 1 || !owner
|| native.permissionTools.some(tool => tool.id !== owner.nativeToolId && tool.name !== 'Bash')
|| this.opts.granted.has(`request:${owner.requestId}`) || !this.clipped(owner, frame.text)) return false;
this.owner = structuredClone(owner);
this.paints = 0;
return this.repaint();
}
private clipped(owner: NativeFilePermissionRequest, visible: string): boolean {
const target = currentFilePermissionTarget(visible), file = owner.input.file_path;
return target !== null && owner.name === (target.operation === 'edit' ? 'Edit' : 'Write')
&& typeof file === 'string' && path.isAbsolute(file)
&& path.basename(target.filePath) === target.filePath && path.basename(file) === target.filePath
&& !/^ (?:Create|Edit|Overwrite) file$/m.test(visible);
}
private async repaint(): Promise<boolean> {
const mark = await this.opts.session.resizeQuestionViewport!(this.paints === 0 ? 240 : this.paints === 1 ? 480 : 960, this.opts.deadlineAt);
if (mark !== null) { this.inputMark = mark; this.paints++; }
return true;
}
}
/** The PTY renders Markdown without stars and may position spaces via ANSI.
* Keep complete-word bounds and stream order; callers dedupe first observations.
*/
export function observedAutoplanPhases(visible: string): number[] {
return [...visible.matchAll(/\bPhase\s*(\d+(?:\.\d+)?)\s*complete\b/g)]
.map(match => Number(match[1]));
}
export interface AutoplanTranscriptObservation {
file: string | null;
phases: number[];
completedLines: number;
pendingBytes: number;
}
/** Preserve the owned commands needed to investigate a stopped chain before
* its fixture is deleted. This diagnostic cannot change a phase verdict.
*/
export function retainAutoplanFailure(opts: {
configDir: string | null; sessionId: string; observation: unknown;
raw: () => string; visible: () => string; evalDir?: string;
/** Owned question evidence and the last decoded viewport are bounded; raw history stays hashed. */
counting?: { native: ReturnType<typeof readPlanSkillQuestions> | null; dialog: string; events?: QuestionEventSource | null;
frame?: { text: string; rawEnd: number; observedAtMs: number; questionSince: number; viewportInputSince: number } | null };
}): string | null {
try {
const raw = opts.raw();
const visible = opts.visible();
const native = readOwnedClaudeTranscript(opts.configDir, opts.sessionId);
const clip = (text: string, limit = 32_768) => ({
text: text.slice(0, limit), codeUnits: text.length, truncated: text.length > limit,
sha256: createHash('sha256').update(text).digest('hex'),
});
const signature = (value: unknown) => {
const { text: _omitted, ...digest } = clip(JSON.stringify(value) ?? '');
return { type: value === null ? 'null' : Array.isArray(value) ? 'array' : typeof value, ...digest };
};
const safeInput = (name: string, input: any) => ({ ...signature(input),
...(['Read', 'Write', 'Edit'].includes(name) && typeof input?.file_path === 'string' ? { filePath: input.file_path.slice(0, 4096) } : {}),
});
const calls: Array<{ id: string; name: string; input: unknown; timestamp: unknown; cwd: unknown }> = [];
const results = new Map<string, boolean>();
for (const row of native.rows) {
const message = row.message;
if (!Array.isArray(message?.content)) continue;
for (const block of message.content) {
if (row.type === 'user' && message.role === 'user' && block?.type === 'tool_result' && typeof block.tool_use_id === 'string') {
results.set(block.tool_use_id, block.is_error === true);
}
if (row.type === 'assistant' && message.role === 'assistant' && block?.type === 'tool_use'
&& typeof block.id === 'string' && typeof block.name === 'string') {
calls.push({ id: block.id, name: block.name, input: block.input ?? null, timestamp: row.timestamp, cwd: row.cwd });
}
}
}
const pending = calls.filter(call => !results.has(call.id));
const queue = native.rows.map((row, index) => ({ row, index })).filter(({ row }) => row.type === 'queue-operation');
const state = opts.counting?.native;
// The failing reader may not have updated state. Preserve current native
// AUQs and revalidated hook inputs as well as the earlier pending snapshot.
// These later diagnostic reads never supply an answer or change the error.
const observedPending = state?.calls.filter(call => call.result === 'pending') ?? [];
let hookQuestions: QuestionEventCall[] = [];
let hookCompletions: QuestionCompletionEventCall[] = [];
let bashInvocations: BashEventCall[] = [];
let bashCompletions: BashCompletionEventCall[] = [];
let bashRequests: BashPermissionRequestEventCall[] = [];
let hookReadError: ReturnType<typeof clip> | null = null;
if (opts.counting?.events) {
try { hookQuestions = readQuestionEvents(opts.counting.events, { configDir: opts.configDir,
sessionId: opts.sessionId, transcriptFile: native.file });
hookCompletions = readQuestionCompletionEvents(opts.counting.events, { configDir: opts.configDir,
sessionId: opts.sessionId, transcriptFile: native.file });
bashInvocations = readBashEvents(opts.counting.events, { configDir: opts.configDir,
sessionId: opts.sessionId, transcriptFile: native.file });
bashCompletions = readBashCompletionEvents(opts.counting.events, { configDir: opts.configDir,
sessionId: opts.sessionId, transcriptFile: native.file });
bashRequests = readBashPermissionRequestEvents(opts.counting.events, { configDir: opts.configDir,
sessionId: opts.sessionId, transcriptFile: native.file }); }
catch (error) { hookReadError = clip(String(error), 1024); }
}
const bashCandidateIds = [...new Set([...(state?.permissionTools.filter(tool => tool.name === 'Bash').map(tool => tool.id) ?? []),
...bashCompletions.slice().sort((a, b) => b.capturedAtMs - a.capturedAtMs).map(event => event.id),
...bashInvocations.slice().sort((a, b) => b.capturedAtMs - a.capturedAtMs).map(event => event.id)])];
const bashIds = new Set(bashCandidateIds.slice(0, 16));
const selectedBashRequests = bashRequests.filter(request => bashInvocations.some(invoked => bashIds.has(invoked.id)
&& invoked.cwd === request.cwd && JSON.stringify(invoked.input) === JSON.stringify(request.input)));
const nativeQuestions = calls.filter(call => call.name === 'AskUserQuestion');
const candidateIds = [...new Set([...observedPending.slice(-16).map(call => call.id),
...hookCompletions.slice().sort((a, b) => b.capturedAtMs - a.capturedAtMs).map(event => event.id),
...nativeQuestions.slice().reverse().map(call => call.id), ...hookQuestions.map(call => call.id),
...observedPending.map(call => call.id)])];
const questionIds = new Set(candidateIds.slice(0, 16));
const questionBlocks: unknown[] = [];
if (opts.counting) for (const [rowIndex, row] of native.rows.entries()) {
const message = row.message;
if (!Array.isArray(message?.content)) continue;
for (const block of message.content) {
const invocation = row.type === 'assistant' && message.role === 'assistant'
&& block?.type === 'tool_use' && block.name === 'AskUserQuestion' && questionIds.has(block.id);
const result = row.type === 'user' && message.role === 'user'
&& block?.type === 'tool_result' && questionIds.has(block.tool_use_id);
if (!invocation && !result) continue;
questionBlocks.push({ rowIndex, type: row.type, stopReason: message.stop_reason ?? null,
timestamp: typeof row.timestamp === 'string' ? clip(row.timestamp, 256) : null,
cwd: typeof row.cwd === 'string' ? clip(row.cwd, 4096) : null,
blockJson: clip(JSON.stringify(block), 65_536),
...(result && row.toolUseResult !== undefined ? { toolUseResultJson: clip(JSON.stringify(row.toolUseResult), 65_536) } : {}) });
}
}
const fileTarget = opts.counting ? currentFilePermissionTarget(opts.counting.dialog) : null;
const counting = opts.counting ? {
dialog: { ...signature(opts.counting.dialog), currentFileTarget: fileTarget ? { ...fileTarget, filePath: fileTarget.filePath.slice(0, 4096) } : null },
decodedFrame: opts.counting.frame ? { source: 'last-sampled-current-screen', ...clip(opts.counting.frame.text, 65_536),
rawEnd: opts.counting.frame.rawEnd, observedAtMs: opts.counting.frame.observedAtMs,
questionSince: opts.counting.frame.questionSince, viewportInputSince: opts.counting.frame.viewportInputSince } : null,
nativeObserved: state !== null, permissionRequestCapture: state?.permissionRequestCapture ?? null,
permissionToolCount: state?.permissionTools.length ?? null,
permissionTools: state?.permissionTools.slice(-16).map(tool => ({ id: clip(tool.id, 256), name: clip(tool.name, 256),
cwd: typeof tool.cwd === 'string' ? clip(tool.cwd, 4096) : null, input: safeInput(tool.name, tool.input) })) ?? [],
permissionRequestCount: state?.permissionRequests.length ?? null,
permissionRequests: state?.permissionRequests.slice(-16).map(request => ({ requestId: clip(request.requestId, 256),
nativeToolId: request.nativeToolId?.slice(0, 256) ?? null, name: request.name, result: request.result,
capturedAtMs: request.capturedAtMs, cwd: clip(request.cwd, 4096), input: safeInput(request.name, request.input) })) ?? [],
bashEvidence: {
candidateCount: bashCandidateIds.length, candidatesOmitted: Math.max(0, bashCandidateIds.length - 16),
chronology: 'Later owned diagnostic reads; no grant or outcome credit. Pending IDs first, then latest native resolution and invocation observations. Background resolution is not command completion.',
permissionRequestCount: selectedBashRequests.length, permissionRequestsOmitted: Math.max(0, selectedBashRequests.length - 16),
permissionRequests: selectedBashRequests.slice(-16).map(event => ({ requestId: clip(event.requestId, 256),
capturedAtMs: event.capturedAtMs, cwd: clip(event.cwd, 4096), inputJson: clip(JSON.stringify(event.input), 65_536) })),
invocations: bashInvocations.filter(event => bashIds.has(event.id)).map(event => ({ id: clip(event.id, 256),
capturedAtMs: event.capturedAtMs, cwd: clip(event.cwd, 4096), inputJson: clip(JSON.stringify(event.input), 65_536) })),
resolutions: bashCompletions.filter(event => bashIds.has(event.id)).map(event => ({ id: clip(event.id, 256),
hookEventName: event.hookEventName, capturedAtMs: event.capturedAtMs, cwd: clip(event.cwd, 4096),
inputJson: clip(JSON.stringify(event.input), 65_536), responseJson: clip(JSON.stringify(event.response), 65_536) })),
hookReadError,
},
questionEvidence: {
count: observedPending.length, omitted: Math.max(0, observedPending.length - 16),
candidateCount: candidateIds.length, candidatesOmitted: Math.max(0, candidateIds.length - 16),
chronology: 'Earlier counting snapshot; later owned transcript and hook reads are not atomic. Pending IDs have priority, then latest completion observations, latest native AUQs, and remaining hooks. Completion observations do not grant diagnostic answer credit; hook order is not execution order.',
observed: observedPending.filter(call => questionIds.has(call.id)).map(call => ({ id: clip(call.id, 256),
observedResult: call.result, questionsJson: clip(JSON.stringify(call.questions), 65_536),
resultAtRetention: results.has(call.id) ? results.get(call.id) ? 'error' : 'completed'
: nativeQuestions.some(nativeCall => nativeCall.id === call.id) ? 'pending' : 'absent' })),
hookEvents: hookQuestions.filter(event => questionIds.has(event.id)).map(event => ({ id: clip(event.id, 256),
toolName: event.toolName, cwd: clip(event.cwd, 4096), inputJson: clip(JSON.stringify(event.input), 65_536) })),
hookCompletionEvents: hookCompletions.filter(event => questionIds.has(event.id)).map(event => ({ id: clip(event.id, 256),
toolName: event.toolName, capturedAtMs: event.capturedAtMs, cwd: clip(event.cwd, 4096),
inputJson: clip(JSON.stringify(event.input), 65_536), responseJson: clip(JSON.stringify(event.response), 65_536) })),
hookReadError,
nativeBlocks: { count: questionBlocks.length, omitted: Math.max(0, questionBlocks.length - 32), rows: questionBlocks.slice(-32) },
},
queueOperations: { count: queue.length, omitted: Math.max(0, queue.length - 16), rows: queue.slice(-16).map(({ row, index }) => ({
rowIndex: index, operation: typeof row.operation === 'string' ? clip(row.operation, 64) : signature(row.operation),
content: signature(row.content), uuid: typeof row.uuid === 'string' ? clip(row.uuid, 256) : null,
timestamp: typeof row.timestamp === 'string' ? clip(row.timestamp, 256) : null,
})) },
} : undefined;
const record = {
schemaVersion: 1, sessionId: opts.sessionId, capturedAt: new Date().toISOString(),
nativeFile: native.file, completedLines: native.completedLines, pendingBytes: native.pendingBytes,
observation: clip(JSON.stringify(opts.observation)),
calls: calls.slice(-16).map(call => ({ id: clip(call.id, 256), name: clip(call.name, 256),
timestamp: clip(String(call.timestamp), 256),
...(counting ? { input: safeInput(call.name, call.input), cwd: typeof call.cwd === 'string' ? clip(call.cwd, 4096) : null } : { inputJson: clip(JSON.stringify(call.input)) }),
result: results.has(call.id) ? results.get(call.id) ? 'error' : 'completed' : 'pending' })),
callCount: calls.length, callsOmitted: Math.max(0, calls.length - 16),
pendingIds: pending.slice(-64).map(call => clip(call.id, 256)),
pendingCount: pending.length, pendingIdsOmitted: Math.max(0, pending.length - 64),
rawTail: { ...(counting ? signature(raw.slice(-65_536)) : clip(raw.slice(-65_536), 65_536)), omittedPrefixCodeUnits: Math.max(0, raw.length - 65_536) }, rawCodeUnits: raw.length,
visibleTail: { ...(counting ? signature(visible.slice(-65_536)) : clip(visible.slice(-65_536), 65_536)), omittedPrefixCodeUnits: Math.max(0, visible.length - 65_536) }, visibleCodeUnits: visible.length,
...(counting ? { counting } : {}),
limits: counting ? 'Diagnostic only. Selected owned AUQ/Bash hook inputs/results and the last sampled decoded viewport are retained with explicit clipping. The viewport keeps its own observation/input epochs; it is not resampled at retention or proof of current ownership. Other native inputs, queue content and raw/flattened history stay hashed. Native thinking and unrelated/foreign results are omitted. Later evidence cannot change the observation, grant input or establish completion.' : 'Diagnostic only. Pending tools are not proof of a permission prompt or a failed command. Native thinking, signatures and tool results are omitted; clipped command inputs remain incomplete evidence.',
};
const evalDir = opts.evalDir ?? process.env.GSTACK_EVAL_DIR;
if (!evalDir) throw new Error('GSTACK_EVAL_DIR is not configured');
const category = counting ? 'plan-counting' : 'autoplan-chain';
const directory = path.join(evalDir, category);
fs.mkdirSync(directory, { recursive: true, mode: 0o700 });
const file = path.join(directory, `${opts.sessionId}.json`);
const serialized = JSON.stringify(record, null, 2) + '\n';
if (Buffer.byteLength(serialized) > 8_388_608) throw new Error('Autoplan diagnostic exceeds its 8 MiB bound');
fs.writeFileSync(file, serialized, { flag: 'wx', mode: 0o600 });
console.error(`[${category}] failure diagnostic: ${file}`);
return file;
} catch (error) {
try { console.error(`Autoplan failure diagnostic could not be retained: ${String(error).slice(0, 1024)}`); } catch { /* preserve the original outcome */ }
return null;
}
}
function announcedAutoplanPhases(text: string): number[] {
const phases: number[] = [];
let fence: string | null = null;
for (const line of text.split(/\r?\n/)) {
const delimiter = line.match(/^ {0,3}(?:> ?)?(`{3,}|~{3,})/)?.[1];
if (delimiter) {
if (!fence) fence = delimiter;
else if (delimiter[0] === fence[0] && delimiter.length >= fence.length) fence = null;
continue;
}
if (fence) continue;
const announcement = line.match(/^ {0,3}(?:> ?)?(?:\*\*)?Phase +(\d+(?:\.\d+)?) +complete\.(?:\*\*)?(?:\s|$)/);
if (announcement) phases.push(Number(announcement[1]));
}
return phases;
}
/** Read only the UUID pinned at launch, directly below a project directory.
* Tool payloads, user messages, and sidechain responses cannot announce phases.
*/
export function readAutoplanTranscript(configDir: string | null, sessionId: string): AutoplanTranscriptObservation {
const { file, rows, completedLines, pendingBytes } = readOwnedClaudeTranscript(configDir, sessionId);
const phases: number[] = [];
for (const row of rows) {
if (row.type !== 'assistant' || row.message?.role !== 'assistant') continue;
if (!Array.isArray(row.message.content)) continue;
for (const block of row.message.content) {
if (block?.type !== 'text' || typeof block.text !== 'string') continue;
for (const phase of announcedAutoplanPhases(block.text)) {
if (!phases.includes(phase)) phases.push(phase);
}
}
}
return { file, phases, completedLines, pendingBytes };
}
/** Assistant order is authoritative; rendered previews cannot establish order.
* Match its prefix in the PTY stream, ignoring unrelated earlier tool previews.
*/
export function corroboratedAutoplanPhases(assistantPhases: readonly number[], visible: string): number[] {
let count = 0;
for (const phase of observedAutoplanPhases(visible)) {
if (phase === assistantPhases[count]) count++;
}
return assistantPhases.slice(0, count);
}
/** Validate first-observed completion markers in stream order. Poll timestamps
* cannot establish order: several phases may first appear in the same batch.
*/
export function validateAutoplanPhaseOrder(phases: readonly number[]): void {
const observed = phases.join(' -> ') || '(none)';
if (!phases.includes(1) || !phases.includes(3)) {
throw new Error(`Autoplan requires CEO (1) and Eng (3) completion; observed: ${observed}`);
}
if (phases.at(-1) !== 3) {
throw new Error(`Autoplan Eng (3) must complete last; observed: ${observed}`);
}
const expected = [1, 2, 2.5, 3];
let previous = -1;
for (const phase of phases) {
const position = expected.indexOf(phase);
if (position <= previous) {
throw new Error(`Autoplan completion order must be CEO (1), optional Design (2), optional DX (2.5), Eng (3); observed: ${observed}`);
}
previous = position;
}
}
-472
View File
@@ -1,472 +0,0 @@
import { capturePlanCountQuestion, parseNumberedOptions, planCountPrerequisitePick, planCountQuestionInput, planCountSubmissionInput, type AskUserQuestionFingerprint } from './claude-pty-runner';
import type { NativePlanQuestion, NativePlanQuestionCall, NativePublicToolEvent, PlanCountTranscript } from './plan-count-transcript';
import type { readPendingQuestion } from './plan-count-pending-question';
/** A copied native panel is not actionable after prose or inside a code example. */
function activeSetupPanel(visible: string): boolean {
const lines = visible.replace(/\r+\n?/g, '\n').trimEnd().split('\n');
if (!/^Enter[\t ]+to[\t ]+select[\t ]*·[\t ]*↑\/↓[\t ]+to[\t ]+navigate[\t ]*·[\t ]*Esc[\t ]+to[\t ]+cancel$/i.test(lines.at(-1)?.trim() ?? '')) return false;
const header = lines.findLastIndex(line => /^ {0,3}[☐□][^\n]+$/.test(line));
if (header < 0) return false;
let fence: { char: string; length: number } | undefined;
for (const line of lines.slice(0, header)) {
const match = /^ {0,3}(`{3,}|~{3,})(.*)$/.exec(line);
if (!match) continue;
if (!fence) fence = { char: match[1]![0]!, length: match[1]!.length };
else if (match[1]![0] === fence.char && match[1]!.length >= fence.length && !match[2]!.trim()) fence = undefined;
}
if (fence) return false;
const introduction = lines.slice(0, header).findLast(line => !/^[\t ─━-]*$/.test(line)) ?? '';
return !/\b(?:example|sample|quot(?:e[sd]?|ed)|template|source)\b.*\b(?:panel|menu|choices?|prompt|question|below|following)\b/i.test(introduction);
}
export type AutoplanSetupDecision =
| { kind: 'input'; input: string; signatures: string[] }
| { kind: 'waiting' | 'unrelated' }
| { kind: 'unsupported_setup'; setup: 'routing' | 'prerequisite'; prompt: string;
options: Array<{ index: number; label: string }>; identitySource: 'native-bound' | 'current-native-panel' };
/** A long boxed routing question can retain its title after only the header scrolls away. */
function clippedRoutingTitle(visible: string, question: AskUserQuestionFingerprint): string | null {
const lines = visible.replace(/\r+\n?/g, '\n').trimEnd().split('\n');
const title = /^ {0,3}[│┃][\t ]*(.+)$/.exec(lines[0] ?? '')?.[1];
if (!title || !/^(?:D\s*\d+\s*[—–:-]\s*)?Add\s+(?:gstack\s+)?skill\s+routing\s+rules\s+to\s+CLAUDE\.md\?\s*<gstack-qid:routing-injection>$/i.test(title)) return null;
const cursors = lines.flatMap((line, index) => /❯\s*[1-9]\./.test(line) ? [index] : []);
if (cursors.length !== 1 || !/^ {0,3}❯\s*1\./.test(lines[cursors[0]!]!)) return null;
const before = lines.slice(0, cursors[0]);
if (before.some(line => line.trim() && !/^ {0,3}[│┃](?:[\t ]|$)/.test(line)) ||
before.some(line => /^ {0,3}[│┃][\t ]*(?:`{3,}|~{3,}|>)/.test(line)) ||
(before.join('\n').match(/<gstack-qid/gi)?.length ?? 0) !== 1 ||
/[☐□☒]|←[^\n]*Submit|(?:^|\n)[^\n]*[1-9]\.\s*\[[ ✓✔xX]\]/.test(visible)) return null;
if (!/^Enter[\t ]+to[\t ]+select[\t ]*·[\t ]*↑\/↓[\t ]+to[\t ]+navigate[\t ]*·[\t ]*Esc[\t ]+to[\t ]+cancel$/i.test(lines.at(-1)?.trim() ?? '')) return null;
const rows = lines.slice(cursors[0]).flatMap(line => {
const match = /^ {0,3}(?:❯\s*)?([1-9])\.[\t ]*(\S.*?)\s*$/.exec(line);
return match ? [{ index: Number(match[1]), label: match[2]! }] : [];
});
if (rows.length !== 4 || rows.some((row, index) => row.index !== index + 1) ||
rows[2]!.label !== 'Type something.' || rows[3]!.label !== 'Chat about this' ||
JSON.stringify(rows) !== JSON.stringify(question.options)) return null;
return title;
}
/** Identify a complete current setup panel; absence or stale/partial metadata is insufficient. */
function completeSetupOptions(visible: string, pending?: NativePlanQuestionCall): Array<{ index: number; label: string }> | null {
if (!activeSetupPanel(visible)) return null;
const lines = visible.replace(/\r+\n?/g, '\n').trimEnd().split('\n');
const header = lines.findLastIndex(line => /^ {0,3}[☐□][^\n]+$/.test(line));
const introduction = lines.slice(0, header).findLast(line => !/^[\t ─━-]*$/.test(line)) ?? '';
if (/\b(?:example|sample|quot(?:e[sd]?|ed)|template|source)\b[^\n]*:\s*$/i.test(introduction)) return null;
const panel = lines.slice(header);
if (/[←→☒]|✔\s*Submit/.test(panel[0]!) ||
panel.some(line => /(?:^|\s)[1-9]\.\s*\[[ ✓✔xX]\]/.test(line))) return null;
if (panel.filter(line => /^ {0,3}❯\s*1\./.test(line)).length !== 1 ||
panel.filter(line => /❯\s*[1-9]\./.test(line)).length !== 1) return null;
const rows = panel.flatMap(line => {
const match = /^ {0,3}(?:❯\s*)?([1-9])\.[\t ]*(\S.*?)\s*$/.exec(line);
return match ? [{ index: Number(match[1]), label: match[2]! }] : [];
});
const compact = (text: string) => text.replace(/\s+/g, '').toLowerCase();
if (rows.length < 4 || rows.some((row, index) => row.index !== index + 1) ||
compact(rows.at(-2)!.label) !== 'typesomething.' || compact(rows.at(-1)!.label) !== 'chataboutthis') return null;
const options = rows.slice(0, -2);
if (pending) {
const native = pending.questions[0];
const cursor = panel.findIndex(line => /^ {0,3}❯\s*1\./.test(line));
if (pending.answered || pending.failed || pending.questions.length !== 1 || native?.multiSelect ||
compact(panel[0]!.replace(/^ {0,3}[☐□]/, '')) !== compact(native!.header) ||
!compact(panel.slice(1, cursor).join(' ')).includes(compact(native!.question)) ||
options.length !== native!.options.length || options.some((row, index) => compact(row.label) !== compact(native!.options[index]!.label))) return null;
}
return options;
}
function unsupportedSetup(visible: string, question: AskUserQuestionFingerprint,
setup: 'routing' | 'prerequisite', pending?: NativePlanQuestionCall): AutoplanSetupDecision {
const options = completeSetupOptions(visible, pending);
return options ? { kind: 'unsupported_setup', setup, prompt: question.promptSnippet, options,
identitySource: pending ? 'native-bound' : 'current-native-panel' } : { kind: 'waiting' };
}
/** Pure routing policy; native packet validation still requires every displayed identity. */
function routingSetupActions(question: AskUserQuestionFingerprint, allowTemporarySkip: boolean, knownSetupOffer = false) {
const primary = question.promptSnippet.replace(/^(?:Routing\s*rules|CLAUDE\.md)\s*/i, '').split('?', 1)[0]!;
const prompt = primary.replace(/\s+/g, '');
const options = question.options.map(option => {
const label = option.label.split(/[│┌\r\n]/, 1)[0]!.trim();
// A native label may repeat its menu letter. Remove one corresponding
// marker only for action matching; keep the original display identity.
const marker = /^([A-Z])\)[\t ]+/i.exec(label);
const action = marker && marker[1]!.toUpperCase().charCodeAt(0) - 64 === option.index
? label.slice(marker[0].length) : label;
return { index: option.index, title: action.replace(/\s+/g, '') };
});
const add = options.filter(option => /^Add(?:routingrules(?:toCLAUDE\.md)?|toCLAUDE\.md)(?:\(Recommended\))?$/i.test(option.title));
// Match the declined setup action, not every English label separately:
// No thanks/Skip may stand alone or opt into manual invocation. A manual
// migration, deletion, or unrelated workflow is not the opposed action.
// "Only" limits the same manual action; it does not add a second action.
// Use one whole-label grammar with and without a courtesy/Skip prefix.
const manualAction = /^(?:manual(?:invocation|skills)?|(?:I['’]ll)?invoke(?:skills)?manually)(?:[-–—]?only)?$/i;
// A temporary Skip is the same opposed setup action only on an intact
// two-choice panel. Its description may corroborate manual invocation;
// the routing premise and unique Add action below establish its scope.
const decline = options.filter(option => {
const title = option.title.replace(/\(Recommended\)$/i, '');
if (/^Skipfornow$/i.test(title)) return allowTemporarySkip;
// The action can stand alone or follow a short courtesy ('No thanks').
// Cursor redraws can damage that courtesy while leaving 'invoke skills
// manually' intact. Match the complete action, not the spelling of No;
// arbitrary preceding instructions and extra trailing actions still fail.
const manual = title.replace(/^[a-z]{0,3}thanks[,—–-]/i, '');
if (manualAction.test(manual)) return true;
const prefix = /^(?:Nothanks|Skip)(?:[,—–-])?/i.exec(title);
if (!prefix) return false;
// 'No thanks' can be followed by the same explicit Skip action. Strip
// that decline verb before checking any optional manual-invocation text.
const action = title.slice(prefix[0].length).replace(/^skip(?:[,—–-])?/i, '');
return action === '' || manualAction.test(action);
});
const routingId = /<gstack-qid:routing-injection>/i.test(question.promptSnippet);
const routingPremise = /gstack/i.test(prompt) && /CLAUDE\.md/i.test(prompt) && /skillroutingrules/i.test(prompt);
// A qid can replace the longer premise, but cannot override a question
// about a different target. The Add action and question must agree on
// project setup rather than a product routing or taste decision.
const claudeTarget = /CLAUDE\.md/i.test(prompt);
const quotedPremise = /\b(?:plan|spec|document)\s+(?:quotes?|cites?|references?)\b/i.test(primary);
if (!claudeTarget || quotedPremise || (!knownSetupOffer && !routingId && !routingPremise)) return null;
return { add, decline };
}
/** A direct setup offer can carry a decision brief without changing its actions. */
function contextualPacketSetup(question: NativePlanQuestion, pending: NativePlanQuestionCall) {
if (pending.answered !== false || pending.failed !== false || !pending.sessionId || !pending.toolUseId ||
question.options.length !== 2 || question.options.some(option => !option.description?.trim())) return null;
const ids = [...question.question.matchAll(/<gstack-qid:[a-z0-9-]+>/gi)];
const text = question.question.replace(/<gstack-qid:[a-z0-9-]+>/gi, '').trim();
const split = /^([^?]+\?)([\s\S]+)$/.exec(text);
if (!split || !split[2]!.trim() || split[2]!.includes('?')) return null;
// Strip one presentation label, then match the complete substantive offer.
// A quoted/conditional/adjacent offer cannot borrow another tab's actions.
const offer = split[1]!.replace(/^D[1-9]\d*\s*[—–:-]\s*/, '').replace(/\s+/g, ' ').trim();
const context = [split[2]!.trim(), ...question.options.map(option => option.description!.trim())].join('\n');
if (/(?:^|\n|[.!]\s+)(?:Also|Then)\b|\bonly\s+(?:if|after)\b|\bunless\b|\bprovided\s+that\b/i.test(context) ||
/\b(?:must|need\s+to|have\s+to)\s+(?:run|complete|finish)\s*\/office-hours\b/i.test(context) ||
/(?:\/office-hours|design\s+doc(?:ument)?)\s+(?:is\s+)?(?:required|mandatory)\b/i.test(context) ||
/\breview\s+is\s+(?:forbidden|blocked)\b|\b(?:skip|bypass|omit)\s+(?:(?:the|this|full|standard|entire|CEO|design|DX|engineering)\s+)*review\b/i.test(context)) return null;
const descriptions = question.options.map(option => option.description!.trim().replace(/^[✅❌]\s*/, ''));
const fp: AskUserQuestionFingerprint = { signature: '', observedAtMs: 0, preReview: true,
promptSnippet: `${question.header} ${question.question}`,
options: question.options.map((option, index) => ({ index: index + 1, label: option.label })) };
if (/^Routing(?: rules)?$/i.test(question.header.trim()) &&
/^Add\s+(?:gstack\s+)?(?:skill\s+)?routing\s+rules\s+to\s+CLAUDE\.md\?$/i.test(offer) &&
(!ids.length || ids.length === 1 && ids[0]![0].toLowerCase() === '<gstack-qid:routing-injection>')) {
const actions = routingSetupActions(fp, true, true);
if (!actions || actions.add.length !== 1 || actions.decline.length !== 1 ||
actions.add[0]!.index === actions.decline[0]!.index) return null;
const add = actions.add[0]!.index, decline = actions.decline[0]!.index;
if (/^(?:Do not|Don't|Never|Skip|Decline)\s+(?:add(?:ing)?\s+)?routing\s+rules\b/i.test(descriptions[add - 1]!) ||
/^Add\s+(?:skill\s+)?routing\s+rules\b/i.test(descriptions[decline - 1]!)) return null;
return { kind: 'routing', pick: add };
}
if (!/^(?:Design doc|Prerequisites?(?: doc)?)$/i.test(question.header.trim()) || ids.length ||
!/^Run\s*\/office-hours\s+(?:now|first)(?:\s+for\s+(?:a|the)\s+design\s+doc(?:ument)?)?\?$/i.test(offer)) return null;
const run = question.options.findIndex(option => /^Run\s*\/office-hours\s*(?:now|first)(?:\s*\(recommended\))?$/i.test(option.label));
const skip = question.options.findIndex(option => /^Skip\s*[—–-]\s*(?:proceed\s+with\s+)?standard\s+review(?:\s*\(recommended\))?$/i.test(option.label));
if (run < 0 || skip < 0 || run === skip ||
/^(?:Skip|Don't|Do not)\b/i.test(descriptions[run]!) ||
/^Run\s*\/office-hours\b/i.test(descriptions[skip]!)) return null;
// Corroborate the selected label with its short action clause; the rest
// of the description may explain tradeoffs without changing that action.
const sentence = /^([^.!?]+)([.!?]|$)/.exec(descriptions[skip]!);
const action = sentence?.[1]?.trim() ?? '';
if (sentence?.[2] === '?' || !/^(?:Review\s+(?:starts?|begins?)\s+(?:immediately|now)|(?:Start|Begin)\s+(?:the\s+)?(?:standard\s+)?review\s+(?:immediately|now)|Proceed\s+(?:directly\s+)?with\s+(?:the\s+)?standard\s+review)(?:\s+(?:using|with|on)\s+[^.!?]+)?$/i.test(action) ||
/\b(?:after|when|once|until|if|unless|provided|not|never)\b|n['’]t\b/i.test(action)) return null;
return { kind: 'prerequisite', pick: skip + 1 };
}
/** Numbered setup wording may vary; newly admitted forms still consume every description. */
function numberedPacketSetup(question: NativePlanQuestion, pending: NativePlanQuestionCall) {
if (pending.answered !== false || pending.failed !== false || !pending.sessionId || !pending.toolUseId || question.options.length !== 2) return null;
const compact = (value: string | undefined) => (value ?? '').trim().replace(/\s+/g, ' ');
const ids = [...question.question.matchAll(/<gstack-qid:[a-z0-9-]+>/gi)];
const text = compact(question.question.replace(/<gstack-qid:[a-z0-9-]+>/gi, ''))
.replace(/^D[1-9]\d*\s*[—–:-]\s*/, '');
const fp: AskUserQuestionFingerprint = { signature: '', observedAtMs: 0, preReview: true,
promptSnippet: `${question.header} ${question.question}`,
options: question.options.map((option, index) => ({ index: index + 1, label: option.label })) };
const routing = routingSetupActions(fp, true);
if (ids.length === 1 && ids[0]![0].toLowerCase() === '<gstack-qid:routing-injection>' &&
/^Add\s+(?:gstack\s+)?skill\s+routing\s+rules\s+to\s+CLAUDE\.md\?$/i.test(text) &&
routing?.add.length === 1 && routing.decline.length === 1 && routing.add[0]!.index !== routing.decline[0]!.index) {
const add = question.options[routing.add[0]!.index - 1]!;
const decline = question.options[routing.decline[0]!.index - 1]!;
if (/^Appends a skill routing section to CLAUDE\.md so future sessions automatically invoke the right skill(?: \(e\.g\. \/autoplan for reviews, \/ship for deploys\))? without you needing to type the command each time\. One-time setup per project\.$/i.test(compact(add.description)) &&
/^No change to CLAUDE\.md\. You['’]ll continue calling skills yourself with \/skill-name as you do now\.$/i.test(compact(decline.description))) {
return { kind: 'routing', pick: routing.add[0]!.index };
}
}
if (ids.length || !/^No design doc (?:found|exists) for (?:this|the) (?:branch|project)\. Run \/office-hours (?:first|now)(?: to sharpen (?:the|this) review input)?\?$/i.test(text)) return null;
const run = question.options.findIndex(option => /^Run\s*\/office-hours\s*(?:now|first)(?:\s*\(recommended\))?$/i.test(option.label));
const skip = question.options.findIndex(option => /^Skip\s*[,—–-]\s*proceed\s+with\s+(?:standard\s+)?review(?:\s*\(recommended\))?$/i.test(option.label));
if (run < 0 || skip < 0 || run === skip ||
!/^Start the full CEO (?:→|->) Design (?:→|->) DX (?:→|->) Eng review pipeline now using the plan as-is\. (?:Recommended when the plan context is already rich enough\. ?)?(?:\(Recommended\))?$/i.test(compact(question.options[skip]!.description)) ||
!/^Produces a structured problem statement, premise challenge, and explored alternatives before the review\. (?:Takes ~?\d+(?:[–-]\d+)? min\. )?Gives the review sharper, better-grounded input\.$/i.test(compact(question.options[run]!.description))) return null;
return { kind: 'prerequisite', pick: skip + 1 };
}
/** Answer only the known pair of setup offers, using the actual native active tab. */
function setupPacketDecision(visible: string, seen: ReadonlySet<string>, pending: NativePlanQuestionCall): AutoplanSetupDecision {
const waiting: AutoplanSetupDecision = { kind: 'waiting' };
// Validate all questions before touching any tab: a setup question cannot
// lend its policy to an adjacent finding, taste decision or checkbox.
if (pending.questions.length !== 2) return waiting;
const policies = pending.questions.map(question => {
if (question.options.length !== 2) return null;
const ids = [...question.question.matchAll(/<gstack-qid:[a-z0-9-]+>/gi)];
if (ids.length > 1 || (question.question.match(/<gstack-qid/gi)?.length ?? 0) !== ids.length) return null;
const offerText = question.question.replace(/<gstack-qid:[a-z0-9-]+>/gi, '').trim();
const fp: AskUserQuestionFingerprint = { signature: '', observedAtMs: 0, preReview: true,
promptSnippet: `${question.header} ${question.question}`,
options: question.options.map((option, index) => ({ index: index + 1, label: option.label })) };
const routing = routingSetupActions(fp, true);
// A packet must contain only setup. Scope the entire question, including
// any premise, rather than borrowing the first question mark's identity
// while a later sentence asks for an unrelated approval.
const routingOffer = /^(?:gstack\s+works\s+best\s+when\s+(?:your|this|the)\s+project['’]s\s+CLAUDE\.md\s+includes\s+skill\s+routing\s+rules\.\s*)?(?:(?:Should|Can)\s+(?:I|gstack)\s+|Would\s+you\s+like\s+(?:me|gstack)\s+to\s+)?Add\s+(?:(?:gstack\s+)?skill\s+routing\s+rules\s+to\s+(?:this\s+project['’]s\s+)?CLAUDE\.md|(?:skill\s+)?routing\s+rules(?:\s+to\s+CLAUDE\.md)?|them)(?:\s+now)?\?$/i.test(offerText);
if (routingOffer && routing?.add.length === 1 && routing.decline.length === 1 && routing.add[0]!.index !== routing.decline[0]!.index) {
return { kind: 'routing', pick: routing.add[0]!.index };
}
const prerequisite = planCountPrerequisitePick(fp);
const run = question.options.filter(option => /^Run\s*\/office-hours\s*(?:now|first)(?:\s*\(recommended\))?$/i.test(option.label));
// Require an actual prerequisite offer, not a product question that
// happens to mention the absence of an office-hours design document.
const offer = /^No\s+design\s+doc\s+(?:found|exists)(?:\s+for\s+(?:this|the)\s+(?:branch|project))?\.\s*(?:\/office-hours\s+(?:produces|creates|provides)\s+(?:a\s+)?(?:structured\s+)?(?:design\s+doc(?:ument)?|problem\s+statement)(?:,?\s+(?:and\s+)?(?:premise\s+challenge|(?:explored\s+)?alternatives))*(?:\s*[—–-]\s*(?:sharper|better)\s+input\s+for\s+(?:the|this)\s+review)?\.\s*)?(?:Want\s+to\s+|Would\s+you\s+like\s+to\s+)?Run\s+(?:it|\/office-hours)\s+(?:now|first)(?:\s+or\s+proceed\s+with\s+standard\s+review)?\s*\?$/i.test(offerText);
return prerequisite !== null && run.length === 1 && offer ? { kind: 'prerequisite', pick: prerequisite }
: numberedPacketSetup(question, pending) ?? contextualPacketSetup(question, pending);
});
if (policies.some(policy => !policy) || new Set(policies.map(policy => policy!.kind)).size !== 2) return waiting;
const text = visible.replace(/\r+\n?/g, '\n').trimEnd();
const bars = [...text.matchAll(/^ {0,3}←([^\n]*[☐☒][^\n]*)✔\s*Submit\s*→[\t ]*$/gm)];
const footer = /Enter[\t ]+to[\t ]+select[\t ]*·[\t ]*Tab\/Arrow[\t ]+keys[\t ]+to[\t ]+navigate[\t ]*·[\t ]*Esc[\t ]+to[\t ]+cancel$/i;
if (bars.length !== 1 || !footer.test(text)) return waiting;
const bar = bars[0]!;
const tabs = [...bar[1]!.matchAll(/([☐☒])\s*([^☐☒]+)/g)];
const compact = (value: string) => value.replace(/\s+/g, '');
const introduction = text.slice(0, bar.index).split('\n').findLast(line => !/^[\t ─━-]*$/.test(line)) ?? '';
if (/\b(?:example|sample|quot(?:e[sd]?|ed)|template|source)\b[^\n]*:\s*$/i.test(introduction) ||
compact(bar[1]!) !== tabs.map(tab => compact(tab[0])).join('') ||
tabs.length !== pending.questions.length || tabs.some((tab, index) => compact(tab[2]!) !== compact(pending.questions[index]!.header)) ||
/(?:^|\n)[^\n]*[1-9]\.\s*\[[ ✓✔xX]\]/.test(text)) return waiting;
// Project only this actual pane's decoration for the existing full-panel
// validator. The question, labels and native identity remain unchanged.
const project = (header: string) => (text.slice(0, bar.index) + '☐ ' + header +
text.slice(bar.index + bar[0].length).replace(footer, 'Enter to select · ↑/↓ to navigate · Esc to cancel'))
.replace(/(^|\n)[\t ]*[│┃][\t ]?/g, '$1');
if (!activeSetupPanel(project('Setup packet'))) return waiting;
const packetKey = 'autoplan-setup-packet:' + JSON.stringify({sessionId:pending.sessionId,toolUseId:pending.toolUseId,questions:pending.questions});
const choiceKey = (index: number) => `${packetKey}:choice:${index}:${policies[index]!.pick}`;
const submitKey = packetKey + ':submit';
if (planCountSubmissionInput(text) === '\r') {
// Checked tabs are corroboration. Only choices this caller actually
// sent for this same native packet can authorize its final submission.
const options = parseNumberedOptions(text);
if (!/Ready\s+to\s+submit\s+your\s+answers\?\s*❯\s*1\./.test(text) ||
options.length !== 2 || options[0]?.index !== 1 || options[0]?.label !== 'Submit answers' ||
options[1]?.index !== 2 || options[1]?.label !== 'Cancel' || seen.has(submitKey) ||
tabs.some((tab, index) => tab[1] !== '☒' || !seen.has(choiceKey(index)))) return waiting;
return { kind: 'input', input: '\r', signatures: [submitKey] };
}
const captured = new Set(seen);
const fp = capturePlanCountQuestion(text, captured, 0, true, pending);
const index = fp?.nativeQuestionIndex;
if (!fp || fp.nativeCall !== pending || index === undefined || tabs[index]?.[1] !== '☐' || seen.has(choiceKey(index))) return waiting;
const question = pending.questions[index]!;
if (!completeSetupOptions(project(question.header), { ...pending, questions: [question] })) return waiting;
return { kind: 'input', input: planCountQuestionInput(text, fp, policies[index]!.pick),
signatures: [...captured].filter(signature => !seen.has(signature)).concat(choiceKey(index)) };
}
/** Pure classification: only the caller that sends input commits returned identities. */
export function autoplanSetupDecision(visible: string, seen: ReadonlySet<string>, pending?: NativePlanQuestionCall): AutoplanSetupDecision {
if (pending && (pending.answered || pending.failed || !pending.questions.length || pending.questions.some(question => question.multiSelect))) return { kind: 'waiting' };
if (pending && pending.questions.length > 1) return setupPacketDecision(visible, seen, pending);
// A visible packet without its complete native metadata cannot prove that
// its other tabs are setup. Wait for persistence instead of guessing.
if (/←[^\r\n]*[☐☒][^\r\n]*✔\s*Submit\s*→|Enter\s*to\s*select\s*·\s*Tab\/Arrow\s*keys\s*to\s*navigate/.test(visible)) return { kind: 'waiting' };
// Box borders are terminal decoration, not part of an untagged native
// question's wrapped text. Keep its full content for identity matching.
const display = visible.replace(/(^|[\r\n])[\t ]*[│┃][\t ]?/g, '$1');
// A rejected/mismatched menu has received no input. Keep its identities
// available when the actual native call arrives after the visible prompt.
const captured = new Set(seen);
const question = capturePlanCountQuestion(display, captured, 0, true, pending);
if (!question) return { kind: 'waiting' };
// Recover only a still-visible direct routing title from this complete
// boxed native panel. A present native call keeps its existing binding.
const clippedTitle = !pending ? clippedRoutingTitle(visible, question) : null;
if (clippedTitle) question.promptSnippet = clippedTitle;
const answered = (input: string | null): AutoplanSetupDecision => input === null
? { kind: 'waiting' }
: { kind: 'input', input, signatures: [...captured].filter(signature => !seen.has(signature)) };
const prerequisite = planCountPrerequisitePick(question);
const prerequisitePrompt = /\/office-hours/i.test(question.promptSnippet) &&
/(?:no\s*design\s*doc|produce\s*a\s*design\s*doc)/i.test(question.promptSnippet);
if (prerequisitePrompt) {
// The canonical offer has two opposed choices. A known native call must
// match this one-question panel; no mixed packet or substantive choice
// may borrow its skip. Before JSONL flushes, require the complete native
// single-select UI and the existing prerequisite premise/action guard.
const choices = question.options.filter(option =>
!/^(?:Type something\.|Chat about this)$/.test(option.label));
if (choices.length !== 2 || (pending &&
(!question.nativeCall || pending.questions.length !== 1 || pending.questions[0]?.multiSelect))) return { kind: 'waiting' };
if (prerequisite === null) {
// Mentioning a missing design doc or /office-hours in a product/taste
// question does not make it setup. Independently identify an offer to
// run that prerequisite even when its opposed skip is unsupported.
const run = choices.filter(({ label }) => /^Run\s*\/office-hours\s*(?:now|first)(?:\s*\(recommended\))?$/i.test(label));
// UI fingerprints abbreviate long questions; inspect the current
// question before its cursor, with full-panel validation below.
const beforeOptions = display.slice(0, display.search(/❯\s*1\./));
const offer = /\brun\s*\/office-hours\s*(?:now|first)\s*\?(?:\s*<gstack-qid:[^>]+>)?\s*$/i.test(beforeOptions);
return run.length === 1 && offer ? unsupportedSetup(display, question, 'prerequisite', pending) : { kind: 'unrelated' };
}
const input = planCountQuestionInput(display, question, prerequisite);
if (!/^[1-9]\d*$/.test(input ?? '') || !activeSetupPanel(display)) return { kind: 'waiting' };
return answered(input);
}
// The model rephrases the setup question's closing sentence. Its routing
// identity/premise and two opposed setup actions establish what is being
// asked; an exact "Add them now?" sentence is not a stable interface.
// Keep the actual question/premise separate from its header and later ELI10
// prose, which may mention CLAUDE.md even on an unrelated question.
const temporarySkipPanel = question.options.some(option => /^Skip\s*for\s*now(?:\s*\(Recommended\))?$/i.test(option.label))
? completeSetupOptions(display, pending) : null;
const actions = routingSetupActions(question, temporarySkipPanel?.length === 2);
if (!actions) return { kind: 'unrelated' };
// Lettered labels are a new action presentation, not permission to use
// the older damaged-option fallback. Match the complete original menu.
if (question.options.some(option => /^[A-Z]\)[\t ]+/i.test(option.label.trim())) &&
completeSetupOptions(display, pending)?.length !== 2) return { kind: 'waiting' };
const { add, decline } = actions;
if (pending && (!question.nativeCall || pending.questions.length !== 1 || pending.questions[0]?.multiSelect)) return { kind: 'waiting' };
// An intact Add-to-CLAUDE.md action identifies this setup offer even if
// its opposed decline is unsupported. A qid or premise alone must not
// turn substantive/ambiguous choices into an early setup failure.
if (add.length !== 1) return { kind: 'waiting' };
if (decline.length !== 1 || add[0]!.index === decline[0]!.index) return unsupportedSetup(display, question, 'routing', pending);
// Newly admitted shorthand still needs the complete two-choice setup
// panel; a qid alone cannot lend it stale or mismatched native metadata.
if (/manualskills/i.test(decline[0]!.title) && completeSetupOptions(display, pending)?.length !== 2) return { kind: 'waiting' };
// The verified clipped panel has the same native numeric shortcut. Do
// not queue Enter behind it when the single-select header is offscreen.
return answered(clippedTitle ? String(add[0]!.index) : planCountQuestionInput(display, question, add[0]!.index));
}
/** Compatibility wrapper: preserve the existing input-only API. */
export function autoplanRoutingSetupInput(visible: string, seen: Set<string>, pending?: NativePlanQuestionCall): string | null {
const decision = autoplanSetupDecision(visible, seen, pending);
if (decision.kind !== 'input') return null;
for (const signature of decision.signatures) seen.add(signature);
return decision.input;
}
/** A long final gate may scroll its header away. This proves a wait, never an answer. */
function croppedFinalApprovalPanel(visible: string, call: NativePlanQuestionCall): boolean {
const question = call.questions[0]!;
if (!/^Final (?:approval )?gate$/i.test(question.header) ||
!/^(?:D\s*\d+\s*[—–:-]\s*)?Final approval(?: gate)?\s*:[^\n]*\?$/i.test(question.question.split('\n')[0]!) ||
question.options.length < 2 || question.options.length > 7) return false;
const lines = visible.replace(/\r+\n?/g, '\n').trimEnd().split('\n');
const footer = 'Enter to select · ↑/↓ to navigate · Esc to cancel';
if (lines.at(-1)?.trim() !== footer ||
/[☐□☒]|✔\s*Submit|←|(?:^|\n)[^\n]*[1-9]\.\s*\[[ ✓✔xX]\]/.test(visible)) return false;
const rowPattern = /^ {0,3}(❯\s*)?([1-9])\.[\t ]+(\S.*?)\s*$/;
const rows = lines.flatMap((line, at) => {
const match = rowPattern.exec(line);
return match ? [{ at, cursor: !!match[1], index: Number(match[2]), label: match[3]! }] : [];
});
if (rows.length !== question.options.length + 2 || rows.some((row, i) => row.index !== i + 1) ||
!rows[0]!.cursor || rows.slice(1).some(row => row.cursor) ||
rows.at(-2)!.label !== 'Type something.' || rows.at(-1)!.label !== 'Chat about this') return false;
const before = lines.slice(0, rows[0]!.at).filter(line => line.trim());
if (!before.length || before.some(line => !/^ {0,3}[│┃][\t ]/.test(line))) return false;
const excerptLines = before.map(line => line.replace(/^ {0,3}[│┃][\t ]?/, ''));
if (excerptLines.some(line => /^\s*(?:`{3,}|~{3,}|>)/.test(line))) return false;
const normalize = (text: string) => text.replace(/\s+/g, ' ').trim();
// The renderer can truncate the last displayed line as well as crop the top.
// Require one contiguous owned excerpt, including at least one complete native
// line. Never assemble disconnected words or borrow a different question.
const excerpt = normalize(excerptLines.join(' ')).replace(/…$/, '');
const native = normalize(question.question), at = native.indexOf(excerpt);
if (at <= 0 || native.indexOf(excerpt, at + 1) !== -1 ||
!question.question.split('\n').slice(1).some(line => normalize(line) && excerpt.includes(normalize(line)))) return false;
for (const [i, option] of question.options.entries()) {
if (!option.label.trim() || normalize(rows[i]!.label) !== normalize(option.label)) return false;
const description = lines.slice(rows[i]!.at + 1, rows[i + 1]!.at);
if (description.some(line => line.trim() && !/^ {4,}\S/.test(line)) ||
normalize(description.join(' ')) !== normalize(option.description ?? '')) return false;
}
// Only native footer decoration may follow the two utility rows.
const decoration = (line: string) => /^[\t ─━-]*$/.test(line);
return lines.slice(rows.at(-2)!.at + 1, rows.at(-1)!.at).every(decoration) &&
lines.slice(rows.at(-1)!.at + 1, -1).every(decoration);
}
/** Identify a remaining native human wait. The caller must treat it as failure, never phase credit. */
export function autoplanBlockingQuestionBoundary(visible: string, context: {
commandStartedAt: number; viewportCapturedAt: number;
/** Both projections must come from the same owned readPlanCountTranscript poll. */
transcript: PlanCountTranscript; publicTools: NativePublicToolEvent[];
/** Only the current return from the owned, post-command readPendingQuestion. */
pending?: ReturnType<typeof readPendingQuestion>;
}): { sessionId: string; toolUseId: string; source: 'native' | 'pre_tool_use' } | null {
const {transcript, commandStartedAt, viewportCapturedAt, publicTools} = context;
if (transcript.status !== 'ready' || !Number.isFinite(commandStartedAt) ||
!Number.isFinite(viewportCapturedAt) || commandStartedAt < 0 || viewportCapturedAt < commandStartedAt) return null;
const sessions = new Set([...transcript.calls.map(call => call.sessionId),
...transcript.assistantMessages.map(message => message.sessionId)]);
if (sessions.size !== 1 || ![...sessions][0]) return null;
const unanswered = transcript.calls.filter(call => !call.answered && !call.failed);
if (unanswered.length > 1) return null;
const call = unanswered[0] ?? context.pending;
if (!call || !sessions.has(call.sessionId) || !call.toolUseId || call.answered !== false ||
call.failed !== false || call.questions.length !== 1 || call.questions[0]!.multiSelect) return null;
const identity = (questions: unknown): string | null => Array.isArray(questions) && questions.every(q =>
q && typeof q.header === 'string' && typeof q.question === 'string' && Array.isArray(q.options) &&
q.options.every((o: any) => o && typeof o.label === 'string' &&
(o.description === undefined || typeof o.description === 'string')))
? JSON.stringify(questions.map(q => [q.header,q.question,q.multiSelect ?? false,
q.options.map((o: any) => [o.label,o.description ?? ''])])) : null;
if (identity(call.questions) === null) return null;
const uses = publicTools.filter(event => event.sessionId === call.sessionId && event.toolUseId === call.toolUseId && event.kind === 'use');
if (publicTools.some(event => event.sessionId === call.sessionId && event.toolUseId === call.toolUseId && event.kind === 'result')) return null;
let source: 'native' | 'pre_tool_use';
if (unanswered.length) {
const use = uses[0], at = Date.parse(use?.timestamp ?? '');
if (uses.length !== 1 || use?.name !== 'AskUserQuestion' || !Number.isFinite(at) ||
at < commandStartedAt || at > viewportCapturedAt || !Array.isArray(use.input?.questions) ||
identity(use.input.questions as NativePlanQuestion[]) !== identity(call.questions)) return null;
source = 'native';
} else {
// The reader has already checked cwd/config, parent session, timestamp,
// recorder poison/lock state and absence of a published result or call.
if (context.pending?.source !== 'pre_tool_use' || uses.length ||
transcript.calls.some(row => row.sessionId === call.sessionId && row.toolUseId === call.toolUseId)) return null;
source = 'pre_tool_use';
}
const display = visible.replace(/(^|[\r\n])[\t ]*[│┃][\t ]?/g, '$1');
const headerAt = display.search(/^ {0,3}[☐□]/m);
if (headerAt < 0) {
// Crop recovery is limited to an already published native use. The pending
// hook route still requires its original complete current panel.
if (source !== 'native' || !croppedFinalApprovalPanel(visible, call)) return null;
} else if (/^(?:Source|Example|Quoted|Historical|Template)\b[^\n]*:/im.test(display.slice(0, headerAt)) ||
!completeSetupOptions(display, call)) return null;
return {sessionId:call.sessionId,toolUseId:call.toolUseId,source};
}
-14
View File
@@ -1,14 +0,0 @@
import path from 'node:path';
/** Rebase recorded Unix paths without inserting raw backslashes into JSON. */
export function capturedPathRebaser(replacements: Array<[string, string]>) {
const text = (value: string) => replacements.reduce((value, [before, after]) =>
value.replaceAll(before, after.split(path.sep).join('/')), value);
// Translate separators without resolving traversal, redundant separators, or
// relative spelling that an ownership rejection control must still observe.
const file = (value: string) => text(value).split('/').join(path.sep);
const pathKeys = new Set(['cwd', 'config', 'stateRoot', 'file', 'file_path', 'transcriptPath']);
const json = <T>(value: T): T => JSON.parse(JSON.stringify(value), (key, value) =>
typeof value === 'string' ? (pathKeys.has(key) ? file(value) : text(value)) : value);
return { text, file, json };
}
+4 -3
View File
@@ -77,8 +77,9 @@ export interface CarveGuard {
* - 'external' → covered by a dedicated bespoke test (complex fixtures, e.g.
* ship's git/VERSION/CHANGELOG state). The data-driven loop
* skips it; E1 asserts `externalTest` exists instead.
* - 'none' → no behavioral guard; the static invariants still apply.
*/
behavioral: 'plan' | 'prompt' | 'external';
behavioral: 'plan' | 'prompt' | 'external' | 'none';
/** Required when behavioral === 'external': path (repo-relative) to the dedicated test. */
externalTest?: string;
/** Parity: max bytes for the always-loaded skeleton (asserts the carve shrank it). */
@@ -565,8 +566,8 @@ do not launch the downstream skill or open a browser.`,
],
gateAfterStop: 'AskUserQuestion options:',
},
behavioral: 'external',
externalTest: 'test/skill-e2e-autoplan-chain.test.ts', // phase-complete markers live ONLY in sections — its assertions ARE section-read proof
// The retired skill-e2e-autoplan-chain was its only section-read proof.
behavioral: 'none',
maxSkeletonBytes: 70_000, // Phase-specific outside coverage, native fallback, and harness guard.
minUnionBytes: 85_000, // measured union 86,926
mustContain: ['6 Decision Principles', 'TASTE DECISION', 'USER CHALLENGE', 'consensus', 'Restore Point'],
+1 -1
View File
@@ -52,7 +52,7 @@ const PLAN_MD = [
export function registerCarveSectionCase(skill: string): void {
const guard = CARVE_GUARDS[skill];
if (!guard || guard.behavioral === 'external') throw new Error(`No generic carved-skill case for ${skill}`);
if (!guard || (guard.behavioral !== 'plan' && guard.behavioral !== 'prompt')) throw new Error(`No generic carved-skill case for ${skill}`);
// Keep explicit cost-scoped selection; the free census pins every wrapper.
if (only && only !== guard.skill) return;
-81
View File
@@ -1,81 +0,0 @@
import type { AskUserQuestionFingerprint } from './claude-pty-runner';
import { pickCeoCompletionHandoff } from './ceo-completion-handoff';
import { findCeoModeOption } from './ceo-mode-option';
/** Follow the offered recommendation only in the native pre-review approach menu. */
export function pickCeoRecommendedApproach(fp: AskUserQuestionFingerprint): number | null {
const call = fp.nativeCall;
if (!fp.preReview || !call || call.answered !== false || call.failed !== false ||
fp.signature !== `${call.sessionId}:${call.toolUseId}` || call.questions.length !== 1 ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0)) return null;
const q = call.questions[0]!;
if (q.multiSelect || q.options.length < 2 || !/^Approach$/i.test(q.header.trim())) return null;
const ids = [...q.question.matchAll(/<gstack-qid:\s*([a-z0-9-]+)\s*>/gi)];
if (ids.length !== 1 || (q.question.match(/<gstack-qid/gi)?.length ?? 0) !== 1) return null;
const qid = ids[0]![1]!.toLowerCase();
const decision = /^D\s*([1-9]\d*)\s*[—–:-]\s*/i.exec(q.question.trim());
const approachId = /^plan-ceo(?:-review)?-approach(?:-selection)?(?:-d([1-9]\d*))?$/.exec(qid);
// Native routing ids may carry the explicit decision number. An id for a
// different decision cannot borrow this question's recommendation policy.
const planApproachId = approachId !== null &&
(!approachId[1] || approachId[1] === decision?.[1]);
const question = q.question.trim().replace(/^D\s*\d+\s*[—–:-]\s*/i, '');
// Native approach menus vary their routing id and use/follow wording. Match
// the selector's structure, retaining the explicit recommendation below.
const planApproach = planApproachId &&
/^Which implementation approach should this plan (?:use|follow)\?(?:\s|$)/i.test(question);
const testApproach = qid === 'plan-ceo-review-impl-approach' &&
/^Which implementation approach for the [a-z_$][\w$]*(?:\.[a-z_$][\w$]*)*\(\) tests\?(?:\s|$)/i.test(question);
// A named component can be the subject instead of "this plan". Consume the
// complete direct question; the routing id alone cannot authorize a choice.
const directQuestion = question.replace(/\s*<gstack-qid:[^>]+>\s*$/i, '').trim();
const component = /^Which implementation approach for (?:the|this) ((?:[a-z_$][\w$.-]*\s+){0,5})(?:handler|endpoint|service|module|component|adapter|client|worker|pipeline|integration)\?$/i.exec(directQuestion);
const componentApproach = planApproachId &&
component !== null && !/\b(?:and|or|then)\b/i.test(component[1]!);
if (!planApproach && !testApproach && !componentApproach) return null;
if (fp.options.length !== q.options.length || !fp.options.every((option, i) =>
option.index === i + 1 && option.label === q.options[i]!.label)) return null;
const labels = q.options.map(option => option.label.trim());
if (new Set(labels).size !== labels.length) return null;
const recommended = labels.map((label, i) => ({ label, index: i + 1 })).filter(({ label }) =>
/\s\(Recommended\)\s*$/i.test(label) &&
(label.match(/\brecommended\b/gi)?.length ?? 0) === 1 &&
!/\b(?:not|never)\s*\(recommended\)/i.test(label));
return recommended.length === 1 ? recommended[0]!.index : null;
}
/** Fixed-count fixtures review existing defects without opting into expansions. */
function pickCeoCountMode(fp: AskUserQuestionFingerprint): number | null {
const call = fp.nativeCall;
if (!fp.preReview || !call || call.answered !== false || call.failed !== false ||
fp.signature !== `${call.sessionId}:${call.toolUseId}` || call.questions.length !== 1 ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0)) return null;
const q = call.questions[0]!;
if (q.multiSelect || !/^(?:Review )?mode$/i.test(q.header.trim()) || q.options.length !== 4 ||
fp.options.length !== 4 || !fp.options.every((option, i) =>
option.index === i + 1 && option.label === q.options[i]!.label)) return null;
const ids = [...q.question.matchAll(/<gstack-qid:\s*([a-z0-9-]+)\s*>/gi)];
if (ids.length !== 1 || (q.question.match(/<gstack-qid/gi)?.length ?? 0) !== 1 ||
!/^(?:plan-)?ceo-(?:review-)?mode(?:-selection)?$/i.test(ids[0]![1]!)) return null;
const question = q.question.trim().replace(/^D\s*[1-9]\d*\s*[—–:-]\s*/i, '')
.replace(/\s*<gstack-qid:[^>]+>\s*$/i, '');
if (!/^Which review mode should I (?:use|apply)(?: for this (?:test coverage )?plan)?\?$/i.test(question)) return null;
try {
// Require all four modes exactly once. Reuse the mode-routing parser so
// displayed option order and side-panel text cannot pick a different mode.
const positions = (['HOLD SCOPE', 'SCOPE EXPANSION', 'SELECTIVE EXPANSION', 'SCOPE REDUCTION'] as const)
.map(mode => findCeoModeOption(fp.options, mode));
return positions.every(position => position !== null) && new Set(positions).size === 4
? positions[0]! : null;
} catch {
return null;
}
}
/** Preserve the existing manual handoff and all other caller/default choices. */
export function pickCeoCountQuestion(
fp: AskUserQuestionFingerprint,
activeCapture: AskUserQuestionFingerprint = fp,
): number | null {
return pickCeoCountMode(activeCapture) ?? pickCeoRecommendedApproach(activeCapture) ?? pickCeoCompletionHandoff(fp, activeCapture);
}
-477
View File
@@ -1,477 +0,0 @@
import type { AskUserQuestionFingerprint } from './claude-pty-runner';
/** A native choice may carry the CEO-specific recap beside a generic completion question. */
function closedCeoRecap(description: string): boolean {
const clause = /(?:^|[.!?]\s+)((?:The\s+)?CEO\s+review\b[^.!?]{0,240})(?=[.!?]|$)/i.exec(description)?.[1];
if (!clause || /\b(?:if|unless|until|once|when|after|not|never)\b|n['’]t\b/i.test(clause)) return false;
return /\b(?:all(?:\s+(?:gaps?|issues?|findings?))?(?:\s+(?:are|were))?\s+resolved|(?:no|0)\s+unresolved\s+(?:decisions|gaps|issues|findings))(?=\s*\)?\s*(?:;|$))/i.test(clause);
}
/** Past-tense resolution can close a native next-review recap without the word "complete". */
function resolvedCeoRecap(description: string): boolean {
const clause = /(?:^|[.!?]\s+)((?:(?:This|The)\s+)?CEO\s+review\s+resolved\s+[^.!?;]{1,180}\b(?:bugs|gaps|issues|findings))(?=\s*(?:[.!?;]|$))/i.exec(description)?.[1];
return Boolean(clause && !/\b(?:if|unless|until|once|when|after|not|never|some|most|partially|only|of|but|several|few)\b|n['’]t\b/i.test(clause));
}
/** A closed-review declaration plus one direct navigation query, even when its recap follows it. */
function closedReviewNavigation(declaration: string, context: string): boolean {
const question = declaration.replace(/<gstack-qid:[^>]+>/gi, '');
return /^CEO review (?:is )?(?:complete|done|cleared|clean)[.!](?:\s|$)/i.test(question) &&
/(?:^|[.!]\s+)What(?:['’]s)? next\?(?:\s|$)/i.test(question) && closedNavigationContext(context);
}
/** The metadata recap is native question text, not a new substantive choice. */
function isMetadataNavigationQuestion(declaration: string): boolean {
return /^What(?:['’]s|\s+is)\s+(?:the\s+)?next(?:\s+(?:step|review))?\s+after\s+(?:this|the)\s+CEO\s+review\?\s*$/i.test(declaration.trim().split('\n')[0]!);
}
function metadataClosedReviewNavigation(declaration: string, context: string): boolean {
const question = declaration.replace(/<gstack-qid:[^>]+>/gi, '').trim();
return isMetadataNavigationQuestion(question) &&
/^[ \t]{0,3}ELI10:\s*(?:The\s+)?CEO\s+review\s+(?:is\s+)?(?:complete|cleared|clean|done(?:\s+and\s+clear(?:ed)?)?)[.!](?:\s|$)/im.test(question) &&
!/`{3}|~{3}|(?:^|[.!?]\s+)[ \t]*>|\b(?:example|quoted source)\s*:/im.test(context) &&
!/\b(?:incomplete|unfinished)\b|\b(?:review|decisions|findings|issues|gaps)\b[^.!?\n]{0,60}\b(?:not|never)\b|\b(?:isn['’]t|aren['’]t|wasn['’]t|weren['’]t)\b/i.test(context) &&
!/(?:^|[.!?;:]\s+|\b(?:proceed to|continue to|should|must|will|need to|can|could|would|may|might)\s+)(?:(?:please|first|then|also)\s+)*(?:add|fix|repair|implement|resolve|decide)\b/im.test(context) &&
closedNavigationContext(context);
}
/** Scope/risk explanations can contain "if" and "not" without reopening CEO work. */
function explainedMetadataNavigation(declaration: string, descriptions: string[]): boolean {
const question = declaration.replace(/<gstack-qid:[^>]+>/gi, '').trim();
if (!isMetadataNavigationQuestion(question)) return false;
const sentences = [question.split('\n').slice(1).join('\n'), ...descriptions]
.flatMap(text => text.trim().split(/[.!](?:\s+|$)/).map(sentence => sentence.trim()).filter(Boolean));
const resolved = /^(?:\d+|one|two|three|four|five|six|seven|eight|nine|ten) assertion spec gaps were caught and resolved$/i;
const noDesignScope = /^No UI scope was detected, so a design review is not needed$/i;
const stakes = /^Stakes if we pick wrong: skipping the eng review means shipping without an architecture \+ code quality pass$/i;
// Validate every whole sentence before discounting the two inert phrases.
// Additional repair, conditional closure, or a different review denial is
// substantive even when it follows a valid metadata heading or recap.
if (sentences.filter(sentence => resolved.test(sentence)).length !== 1 ||
!sentences.every(sentence => resolved.test(sentence) || noDesignScope.test(sentence) || stakes.test(sentence) ||
/^ELI10: The CEO review is (?:done|complete|cleared|clean)$/i.test(sentence) ||
/^The plan is now ready for the Eng Review, which is the required gate before shipping$/i.test(sentence) ||
/^For test code this is lower risk than production code, but the eng review also validates that the test infrastructure is used correctly$/i.test(sentence) ||
/^Recommendation: [A-Z] because eng review is the required shipping gate, and this plan is ready for it$/i.test(sentence) ||
/^Required gate$/i.test(sentence) ||
/^Validates architecture, test infrastructure usage, code quality, and that the \d+-test plan will be implementable without hidden issues$/i.test(sentence) ||
/^Proceed to implementation without the eng review$/i.test(sentence) ||
/^Lower confidence that the test infrastructure is wired correctly, but acceptable for low-risk test coverage work$/i.test(sentence))) return false;
const normalized = [question.split('\n')[0]!, ...sentences
.filter(sentence => !noDesignScope.test(sentence))
.map(sentence => stakes.test(sentence) ? sentence.replace(/^Stakes if we pick wrong:/i, 'Stakes:') : sentence)]
.join('\n');
return metadataClosedReviewNavigation(question, normalized);
}
/** A direct Eng/manual choice can put its unconditional CEO recap in a native description. */
function describedEngNavigation(question: string, descriptions: string[], context: string): boolean {
if (!/^Run\s+\/plan-eng-review\s+(?:next|now)\s*\((?:the\s+)?required(?:\s+shipping)?\s+gate\),?\s+or\s+handle\s+reviews\s+manually\?$/i.test(question)) return false;
const recap = /^(?:The\s+)?CEO\s+review\s+is\s+(?:clear|complete|cleared|clean|done)(?:\s+but\s+eng\s+review\s+is\s+the\s+(?:required\s+)?shipping\s+gate)?$/i;
const topics = String.raw`(?:test isolation|factory patterns|test coverage|architecture|dependencies)`;
const reviewExplanation = new RegExp(String.raw`^(?:Validates|Checks|Reviews)\s+${topics}(?:,\s+${topics})*(?:,?\s+and\s+(?:${topics}|confirms no hidden dependencies))?$`, 'i');
const sentences = descriptions.flatMap(description => description.trim().split(/[.!](?:\s+|$)/).map(sentence => sentence.trim()).filter(Boolean));
const closedRecap = sentences.some(sentence => recap.test(sentence));
// Every sentence must explain this closed handoff. Arbitrary prose after
// a valid recap could add work (including verbs no blacklist anticipates).
if (!sentences.every(sentence => recap.test(sentence) || reviewExplanation.test(sentence) ||
/^Required(?:\s+shipping)?\s+gate\s+before\s+(?:shipping|merging|implementation)$/i.test(sentence) ||
/^(?:You['’]ll|You will)\s+need\s+to\s+run\s+\/plan-eng-review\s+(?:separately\s+)?before\s+(?:merging|shipping)$/i.test(sentence) ||
/^Run\s+\/plan-eng-review\s+(?:next|now|before\s+(?:merging|shipping)|after\s+implementation\s+and\s+before\s+shipping)$/i.test(sentence) ||
/^(?:Fast|Quick|Short)\s+(?:run|review)\s+expected\s+given\s+(?:zero|no|0)\s+CEO\s+findings$/i.test(sentence))) return false;
if (!closedRecap || /`{3}|~{3}|(?:^|\n)[ \t]*>|\b(?:example|quoted source)\s*:/im.test(context) ||
/\b(?:incomplete|unfinished|not|never)\b|n['’]t\b/i.test(context)) return false;
// "Clear" is also a closure claim here; a future condition cannot supply it.
const clearClosure = String.raw`(?:(?:the\s+)?CEO|the)\s+review\s+(?:(?:is|was|becomes?|became|(?:will|would|can|could|may|might)\s+(?:be|become))\s+)?clear`;
if (new RegExp(String.raw`\b(?:once|when|after)\b[^.!?]{0,180}\b${clearClosure}\b|\b${clearClosure}\b[^.!?]{0,100}\b(?:once|when|after)\b`, 'i').test(context)) return false;
return !/(?:^|[.!?;:]\s+|\b(?:proceed to|continue to|should|must|will|need to|can|could|would|may|might)\s+)(?:(?:please|first|then|also)\s+)*(?:add|fix|repair|implement|resolve|decide)\b/im.test(context) &&
closedNavigationContext(context);
}
/** The next sentence may name the required gate with "it" after the Eng query. */
function pronounEngGate(question: string, descriptions: string[], context: string): boolean {
if (!/^(?:The\s+)?CEO\s+review\s+is\s+(?:complete|cleared|clean|done)[.!]\s+Run\s+\/plan-eng-review\s+next\?\s+It(?:['’]s|\s+is)\s+the\s+required(?:\s+shipping)?\s+gate\.$/i.test(question)) return false;
const topics = String.raw`(?:architecture|security|test quality|performance)`;
const covers = new RegExp(String.raw`^Covers\s+${topics}(?:,\s+${topics})*(?:,?\s+and\s+${topics})?$`, 'i');
const sentences = descriptions.flatMap(description => description.trim().split(/[.!](?:\s+|$)/)
.map(sentence => sentence.trim()).filter(Boolean));
return sentences.every(sentence => covers.test(sentence) ||
/^Required\s+gate\s+before\s+shipping$/i.test(sentence) ||
/^This\s+CEO\s+review\s+found\s+no\s+architecture\s+concerns,\s+so\s+eng\s+review\s+should\s+be\s+fast$/i.test(sentence) ||
/^Proceed\s+without\s+the\s+eng\s+review\s+gate$/i.test(sentence) ||
/^You\s+own\s+ensuring\s+correctness\s+before\s+shipping$/i.test(sentence)) &&
closedNavigationContext(context);
}
/** A next-review question may explain completed CEO work only in its choices. */
function describedPostReviewNavigation(question: string, descriptions: string[], context: string): boolean {
if (!/^What(?:['’]s|\s+is)\s+the\s+next\s+review\s+step\s+after\s+(?:this|the)\s+CEO\s+review\?$/i.test(question) ||
/`{3}|~{3}|(?:^|\n)[ \t]*>|\b(?:example|quoted source)\s*:/im.test(context)) return false;
const closed = /^(?:The|This) CEO review resolved all findings, but the eng review validates the approach at a lower implementation level$/i;
const topics = String.raw`(?:[\w-]+ integration|parameterized queries|async [\w-]+ queue)`;
const changedApproach = new RegExp(String.raw`^This CEO review changed the implementation approach \(Approach [A-Z]: ${topics}(?:, ${topics})*\) [—–-] a fresh eng review should validate the new approach before implementation begins$`, 'i');
const sentences = descriptions.flatMap(description => description.trim().split(/[.!](?:\s+|$)/)
.map(sentence => sentence.trim()).filter(Boolean));
// Whole sentences keep extra work out of the recap, including actions that
// an imperative-verb blacklist would miss. Only the next review is offered.
return sentences.filter(sentence => closed.test(sentence)).length === 1 &&
sentences.every(sentence => closed.test(sentence) || changedApproach.test(sentence) ||
/^Eng review is the required shipping gate$/i.test(sentence) ||
/^It covers architecture details, code quality, and test verification$/i.test(sentence) ||
/^Proceed to implementation without the eng review gate$/i.test(sentence) ||
/^Skipping is not recommended for a handler that processes payment webhooks$/i.test(sentence)) &&
closedNavigationContext(context);
}
/** A resolved-gap count may qualify completion before the required next gate. */
function countedCeoNavigation(question: string, descriptions: string[]): boolean {
if (!/^(?:The )?CEO review is complete \(0 critical gaps, [1-9]\d* (?:spec )?gaps resolved\)\. Eng Review is the required shipping gate\. What['’]s next\?$/i.test(question)) return false;
const topics = String.raw`(?:architecture|code quality|tests|performance)`;
const covers = new RegExp(String.raw`^Covers ${topics}(?:, ${topics})*(?:,? and ${topics})?$`, 'i');
return descriptions.flatMap(description => description.trim().split(/[.!](?:\s+|$)/)
.map(sentence => sentence.trim()).filter(Boolean)).every(sentence => covers.test(sentence) ||
/^Required gate before shipping$/i.test(sentence) ||
/^This is a test-only plan so eng review should be fast$/i.test(sentence) ||
/^You manage the eng review yourself$/i.test(sentence) ||
/^The dashboard will show NOT CLEARED until it runs$/i.test(sentence));
}
/** An unconditional CLEAR recap followed by one direct required-Eng query. */
function clearRequiredEngNavigation(question: string, descriptions: string[], context: string): boolean {
if (!/^(?:The\s+)?CEO\s+review\s+is\s+CLEAR\.\s+Eng\s+review\s+is\s+the\s+required\s+shipping\s+gate\s+[—–-]\s+run\s+it\s+next\?$/i.test(question) ||
descriptions.some(description => !description.trim())) return false;
const topics = String.raw`(?:architecture|code quality|tests|performance)`;
const topicsReview = new RegExp(String.raw`^${topics}(?:,\s+${topics})*(?:,?\s+and\s+${topics})?\s+review$`, 'i');
const resolved = /^This\s+CEO\s+review\s+held\s+scope\s+and\s+resolved\s+[1-9]\d*\s+assertion\s+gaps\s+[—–-]\s+eng\s+review\s+verifies\s+the\s+test\s+structure\s+is\s+sound$/i;
const sentences = descriptions.flatMap(description => description.trim().split(/[.!](?:\s+|$)/)
.map(sentence => sentence.trim()).filter(Boolean));
// CLEAR is accepted only with this complete navigation grammar. Do not add
// it to the permissive legacy completion regex or discard appended prose.
return sentences.filter(sentence => resolved.test(sentence)).length === 1 &&
sentences.every(sentence => resolved.test(sentence) || topicsReview.test(sentence) ||
/^Required\s+gate\s+before\s+shipping$/i.test(sentence) ||
/^You\s+manage\s+the\s+review\s+pipeline\s+yourself$/i.test(sentence) ||
/^Note:\s+eng\s+review\s+is\s+required\s+to\s+CLEAR\s+for\s+\/ship$/i.test(sentence)) &&
closedNavigationContext(context);
}
/** A bare next-workflow choice is administration, never proof of completed review. */
function bareEngNavigation(question: string, descriptions: string[]): boolean {
if (!/^run \/plan-eng-review\?$/i.test(question) || descriptions.some(s => !s.trim())) return false;
const sentences = descriptions.flatMap(s => s.trim().split(/\n+|[.!](?:\s+|$)/))
.map(s => s.trim().replace(/^\[[+-]\]\s*/, '')).filter(Boolean);
const approved = /^Proceed directly to implementation with the approved changes from this CEO review$/i;
const gate = /^Eng Review is the required shipping gate$/i;
// Consume the complete offered context. Past findings and already-approved
// changes are recaps; an added remedy or unfinished-review choice is not.
return sentences.some(s => approved.test(s)) && sentences.some(s => gate.test(s)) &&
sentences.every(s => approved.test(s) || gate.test(s) ||
/^It covers architecture depth, code quality, test gaps, and performance [—–-] complementing what this CEO review found$/i.test(s) ||
/^Since this CEO review expanded the plan \(added [a-z0-9_ +/-]{1,120} requirements\), a fresh eng review is especially valuable$/i.test(s) ||
/^Required before shipping; catches implementation issues the plan-level review cannot$/i.test(s) ||
/^This CEO review found critical issues \([a-z0-9_ +/-]{1,80}\) [—–-] eng review will verify the fix approach is architecturally sound$/i.test(s) ||
/^Adds another review session before implementation starts$/i.test(s) ||
/^Faster path to implementation$/i.test(s) ||
/^Eng review is the required shipping gate [—–-] skipping it means less confidence before enabling the feature flag$/i.test(s));
}
/** A completed CEO review may distinguish the still-unrun Eng shipping gate. */
function unrunEngNavigation(fp: AskUserQuestionFingerprint, question: string): number | null {
const call = fp.nativeCall!;
const q = call.questions[0]!;
if (!call.sessionId || !call.toolUseId || call.failed !== false ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0) ||
q.options.length !== 2 || fp.options.length !== 2 ||
!fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label) ||
!/^Next review$/i.test(q.header.trim()) ||
!/^CEO Review is CLEAR\. Eng Review is the required shipping gate and (?:hasn['’]t|has not) run yet\. What(?:['’]s| is) next\?$/i.test(question)) return null;
if (call.answered === false) {
if (call.answers !== undefined || call.answeredAt !== undefined ||
(call.unansweredQuestionIndices !== undefined &&
(call.unansweredQuestionIndices.length !== 1 || call.unansweredQuestionIndices[0] !== 0))) return null;
} else if (call.answered !== true || !Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length) return null;
const labels = q.options.map(o => o.label.trim().replace(/^[A-Z][).]\s*/i, '').replace(/\s*\(recommended\)\s*$/i, '').trim());
const run = labels.findIndex(s => /^Run \/plan-eng-review(?: next| now)?$/i.test(s));
const manual = labels.findIndex(s => /^Skip\s*[—–-]\s*I['’]ll handle reviews manually$/i.test(s));
if (run < 0 || manual < 0 || run === manual) return null;
const topics = String.raw`(?:architecture|code quality|test design|performance|deployment)`;
const runDescription = new RegExp(String.raw`^${topics}(?:,\s+${topics})*(?:,?\s+and\s+${topics})?\s+review\.\s+The required gate before shipping\.\s+Run this before implementation begins to catch any structural issues in how the tests are wired up\.$`, 'i');
// Consume each complete description in its own offered role. The temporal
// qualification is about the next review, not an unfinished CEO decision.
if (!runDescription.test(q.options[run]!.description?.trim() ?? '') ||
!/^Proceed to implementation directly\.\s+You can run \/plan-eng-review later if needed\.\s+Eng Review is required before shipping but not before starting implementation\.$/i.test(q.options[manual]!.description?.trim() ?? '')) return null;
return manual + 1;
}
/** A completed review can explain the cost of skipping its next required gate. */
function explainedRequiredEngNavigation(fp: AskUserQuestionFingerprint, question: string): number | null {
const call = fp.nativeCall!, q = call.questions[0]!;
if (!call.sessionId || !call.toolUseId || call.failed !== false ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0) ||
q.header.trim() !== 'Next step' || q.options.length !== 2 || fp.options.length !== 2 ||
!fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label)) return null;
if (call.answered === false) {
if (call.answers !== undefined || call.answeredAt !== undefined ||
(call.unansweredQuestionIndices !== undefined &&
(call.unansweredQuestionIndices.length !== 1 || call.unansweredQuestionIndices[0] !== 0))) return null;
} else if (call.answered !== true || !Array.isArray(call.unansweredQuestionIndices) ||
call.unansweredQuestionIndices.length || Object.keys(call.answers ?? {}).length !== 1) return null;
const compact = (s: string | undefined) => (s ?? '').replace(/\s+/g, ' ').trim();
const match = /^What(?:['’]s| is) next after this CEO review\? ELI10: The CEO review is done and the plan is CLEARED\. But Eng Review is the required shipping gate [—–-] it covers architecture, test plan rigor, and implementation correctness in more depth\. Running it next locks in the plan before implementation starts\. Stakes if we pick wrong: Skipping eng review means the plan goes to implementation without a required gate check [—–-] leaving architecture and test-correctness gaps unverified\. Recommendation: ([A-Z]) because the dashboard shows Eng Review at 0 runs [—–-] required gate, not yet cleared\. Note: options differ in kind, not coverage [—–-] no completeness score\.$/.exec(compact(question));
if (!match) return null;
const labels = q.options.map(o => o.label.trim().replace(/^[A-Z][).]\s*/, '').replace(/\s*\(Recommended\)$/, ''));
const run = labels.indexOf('Run /plan-eng-review next'), manual = labels.indexOf('Skip — handle reviews manually');
if (run < 0 || manual < 0 || run === manual || !q.options[run]!.label.startsWith(`${match[1]}) `)) return null;
// All question prose and each role-specific description must be closed
// navigation. The risk explanation is not a new CEO repair decision.
if (!/^Required shipping gate\. Covers implementation correctness, test plan rigor, and any architecture concerns\. Takes ~[1-9]\d* minutes\.$/.test(compact(q.options[run]!.description)) ||
compact(q.options[manual]!.description) !== 'Proceed to implementation without the eng review gate. CEO review findings still apply.') return null;
return manual + 1;
}
/** Shared closed-review guards; next-review sequencing is still navigation. */
function closedNavigationContext(context: string): boolean {
const unfinished = context.replace(/\b(?:no|0)\s+unresolved\s+(?:decisions|gaps|issues|findings)\b/gi, '');
// Conditional closure of this review is unfinished work. Sequencing the
// next review after implementation does not reopen the completed CEO review.
const stateVerb = String.raw`(?:is|are|was|were|becomes?|became|(?:will|would|can|could|may|might)\s+(?:be|become))`;
const closure = String.raw`(?:(?:all\s+)?(?:decisions|gaps|issues|findings)\s+(?:${stateVerb}\s+)?resolved|(?:the\s+)?CEO\s+review\s+(?:${stateVerb}\s+)?(?:complete|done|cleared|clean)|the\s+review\s+(?:${stateVerb}\s+)?(?:complete|done|cleared|clean))`;
const conditionalClosure = new RegExp(String.raw`\b(?:once|when|after)\b[^.!?]{0,180}\b${closure}\b|\b${closure}\b[^.!?]{0,100}\b(?:once|when|after)\b`, 'i');
return (context.match(/\?/g)?.length ?? 0) === 1 &&
!/\b(?:unresolved|outstanding|remaining|pending|if|unless|until)\b|\b(?:gap|issue|finding|decision)s?\s+(?:still\s+)?remains?\b|\bstill\s+open\b/i.test(unfinished) &&
!/\bnot\s+(?:all|no|0)\b/i.test(context) &&
!conditionalClosure.test(context) &&
!/(?:^|[.!?;]\s+|\b(?:proceed to|continue to|should|must|will|need to|can|could|would)\s+)(?:(?:please|first|then|also)\s+)*(?:add|fix|implement|resolve|decide)\b/im.test(context);
}
/** A pure next-review menu remains navigation when question tuning is off. */
function sequencedReviewNavigation(fp: AskUserQuestionFingerprint): number | null {
const call = fp.nativeCall!, q = call.questions[0]!;
if (call.failed !== false || !call.sessionId || !call.toolUseId || q.header.trim() !== 'Next review' ||
q.options.length !== 2 || fp.options.length !== 2 || q.multiSelect ||
!fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label) ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0)) return null;
if (call.answered === false) {
if (call.answers !== undefined || call.answeredAt !== undefined ||
(call.unansweredQuestionIndices !== undefined &&
(call.unansweredQuestionIndices.length !== 1 || call.unansweredQuestionIndices[0] !== 0))) return null;
} else if (call.answered !== true || !Array.isArray(call.unansweredQuestionIndices) ||
call.unansweredQuestionIndices.length || Object.keys(call.answers ?? {}).length !== 1) return null;
const lines = q.question.trim().split('\n').map(line => line.trim()).filter(Boolean);
if (!/^D[1-9]\d* [—–-] Which review runs next\?$/.test(lines[0] ?? '')) return null;
// Consume the entire brief, including the displayed option explanations.
// Only the next gate is open; adding a new remedy anywhere rejects this arm.
const grammar = [
/^Project\/branch\/task: [\w.-]+ on [\w./-]+; CEO review of [\w./-]+ is complete and clean \(HOLD SCOPE, 0 critical gaps, [1-9]\d* P1 tasks\)\.$/,
/^ELI10: gstack chains reviews\. The CEO review just settled scope and strategy\. The engineering review is the required gate before shipping: it checks architecture, test design, and code quality in detail\. skip_eng_review is false, so it is still required\. No UI scope was detected, so the design review does not apply here\.$/,
/^Stakes if we pick wrong: skipping eng review leaves the ship gate NOT CLEARED; the plan is small, so the eng review should be quick\.$/,
/^Recommendation: A because eng review is the required gate and the plan now has exact assertions worth a second structured pass on test design\.$/,
/^Note: options differ in kind, not coverage [—–-] no completeness score\.$/,
/^A\) Run \/plan-eng-review next \(recommended\)$/,
/^✅ Clears the required shipping gate on a plan that is small and already decided$/,
/^✅ Gives the three tasks a test-design pass focused on the assertion mechanics \(mock implementation, sleeper record shape\)$/,
/^❌ One more review session before implementation starts \(human ~[1-9]\d* min \/ CC ~[1-9]\d* min\)$/,
/^B\) Skip, handle reviews manually$/,
/^✅ Move straight to implementing T[1-9]\d* to T[1-9]\d* in the real repo$/,
/^✅ No further review time on a three-task change$/,
/^❌ Dashboard verdict stays NOT CLEARED until an eng review is logged$/,
/^Net: gate discipline versus getting to the code faster on a change that is already tightly specified\.$/,
];
if (lines.length !== grammar.length + 1 || !grammar.every((re, i) => re.test(lines[i + 1]!))) return null;
const labels = q.options.map(o => o.label.trim().replace(/^[A-Z]:\s*/, '').replace(/\s*\(recommended\)$/, ''));
const run = labels.indexOf('Run /plan-eng-review next'), manual = labels.indexOf('Skip, manual reviews');
if (run < 0 || manual < 0 || run === manual ||
q.options[run]!.description !== 'Required gate; runs after this plan is approved.' ||
q.options[manual]!.description !== 'Proceed to implementation; eng gate remains open.') return null;
return manual + 1;
}
/** Closed CEO next-review navigation; native terminal/report checks prove completion separately. */
function manualHandoffIndex(fp: AskUserQuestionFingerprint): number | null {
const call = fp.nativeCall;
// The capture path assigns this native identity only after matching the
// active question. UI-only and mismatched pending records cannot steer it.
if (!call || call.failed || fp.signature !== `${call.sessionId}:${call.toolUseId}` || call.questions.length !== 1) return null;
const q = call.questions[0]!;
if (q.multiSelect || q.options.length < 2) return null;
const ids = [...q.question.matchAll(/<gstack-qid:\s*([a-z0-9-]+)\s*>/gi)];
if (ids.length > 1 || (q.question.match(/<gstack-qid/gi)?.length ?? 0) !== ids.length) return null;
const id = ids[0]?.[1]?.toLowerCase();
if (!id) {
const sequenced = sequencedReviewNavigation(fp);
if (sequenced !== null) return sequenced;
}
if (id && !/^(?:plan-ceo-(?:review-)?next-(?:steps?|review)|ceo-review-next-(?:steps?|review)|ceo-next-step-eng-review|ceo-plan-next-steps)$/.test(id)) return null;
const declaration = q.question.replace(/^D\s*\d+\s*[—–:-]\s*/i, '')
.replace(/^next\s+(?:review|steps?)\s*:\s*/i, '');
const gateContext = [q.question, ...q.options.map(option => option.description ?? '')].join('\n');
const explicitCompletion = /(?:^|[.!?]\s+)(?:ELI10:\s*)?(?:The\s+)?CEO\s+review\s+(?:is\s+)?(?:complete|cleared|clean|done(?:\s+and\s+the\s+plan\s+is\s+cleared)?)(?:\s+with\s+0\s+unresolved\s+decisions)?(?=\s*(?:[.!?—–]|$))/i.test(declaration);
const genericCompletion = /(?:^|[.!?]\s+)(?:The\s+)?review\s+(?:is\s+)?(?:complete|cleared|clean|done)(?=\s*(?:[.!?—–]|$))/i.test(declaration);
const questionText = declaration.replace(/<gstack-qid:[^>]+>/gi, '').trim();
const unrunNavigation = id ? unrunEngNavigation(fp, questionText) : null;
if (unrunNavigation !== null) return unrunNavigation;
const explainedNavigation = id ? explainedRequiredEngNavigation(fp, questionText) : null;
if (explainedNavigation !== null) return explainedNavigation;
const recappedNavigation = Boolean(id) &&
/^What(?:['’]s|\s+is)\s+the\s+next\s+(?:steps?|review)\s+after\s+(?:this|the)\s+CEO\s+review\?$/i.test(questionText) &&
q.options.some(option => resolvedCeoRecap(option.description ?? ''));
const unfinished = gateContext.replace(/\b(?:no|0)\s+unresolved\s+(?:decisions|gaps|issues|findings)\b/gi, '');
const describedCompletion = (recappedNavigation || (genericCompletion && q.options.some(option => closedCeoRecap(option.description ?? '')))) &&
!/\b(?:unresolved|outstanding|remains?|remaining|pending)\b/i.test(unfinished) &&
!/(?:^|[.!?;]\s+|\b(?:please|must|need\s+to)\s+)(?:(?:please|first|then|also)\s+)*(?:add|fix|implement|resolve|decide)\b/im.test(gateContext);
const metadataCompletion = Boolean(id) && (metadataClosedReviewNavigation(declaration, gateContext) ||
(call.failed === false && q.options.length === 2 &&
explainedMetadataNavigation(declaration, q.options.map(option => option.description ?? ''))));
if (isMetadataNavigationQuestion(questionText) && /\n[ \t]*ELI10:/i.test(questionText) && !metadataCompletion) return null;
const describedEngCompletion = q.options.length === 2 &&
describedEngNavigation(questionText, q.options.map(option => option.description ?? ''), gateContext);
const describedPostReviewCompletion = !id && q.options.length === 2 &&
describedPostReviewNavigation(questionText, q.options.map(option => option.description ?? ''), gateContext);
const countedCompletion = !id && call.failed === false && q.options.length === 2 &&
countedCeoNavigation(questionText, q.options.map(option => option.description ?? ''));
const clearCompletion = !id && call.failed === false && q.options.length === 2 &&
clearRequiredEngNavigation(questionText, q.options.map(option => option.description ?? ''), gateContext);
const completion = explicitCompletion || describedCompletion || metadataCompletion || describedEngCompletion || describedPostReviewCompletion || countedCompletion || clearCompletion;
const bareNavigation = Boolean(id) && call.failed === false && q.options.length === 2 &&
(fp.nativeQuestionIndex === undefined || fp.nativeQuestionIndex === 0) &&
fp.options.length === 2 && fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label) &&
bareEngNavigation(questionText, q.options.map(o => o.description ?? ''));
const requiredEng = /(?:\bEng(?:ineering)?\s+review|\/plan-eng-review)\b[^.!?]{0,180}\brequired(?:\s+shipping)?\s+gate\b/i.test(gateContext) ||
/\brequired(?:\s+shipping)?\s+gate\s+is\s+(?:an?\s+)?(?:Eng(?:ineering)?\s+review|\/plan-eng-review)\b/i.test(gateContext) ||
pronounEngGate(questionText, q.options.map(option => option.description ?? ''), gateContext);
// These native next-review identities share a closed navigation contract;
// the question or a following recap cannot hide a new repair obligation.
if (id && /^(?:ceo-plan-next-steps|ceo-review-next-(?:steps?|review))$/.test(id) &&
!closedReviewNavigation(declaration, gateContext)) return null;
// A qid alone cannot authorize another fix. The bare navigation arm grants
// no completion credit; native Exit, report freshness and finding floor remain independent.
if (!/^next\s+(?:review|steps?)$/i.test(q.header.trim()) || !(completion || bareNavigation) || !requiredEng) return null;
const labels = q.options.map(o => o.label.trim().replace(/^[A-Z][).]\s*/i, '').replace(/\s*\(recommended\)\s*$/i, '').trim());
if (clearCompletion && !labels.some(label => /^Run\s+\/plan-eng-review(?:\s+(?:next|now))?$/i.test(label))) return null;
const runs = labels.map(label => /^Run\s+\/plan-(?:eng|design)-review(?:\s+(?:next|now))?(?:\s*\(required gate\))?$/i.test(label));
const manual = labels.map(label => /^(?:Skip|Done)\s*[—–-]\s*(?:I['’]ll\s+)?handle\s+(?:reviews\s+)?manually$/i.test(label));
// Deferring the next review until after already-approved implementation is
// navigation too. A new fix/TODO/task choice remains substantive. The picker
// always selects manual, never this implementation route.
const deferred = labels.map((label, i) => /^Implement\s+now,\s+eng\s+review\s+later$/i.test(label) &&
/^Proceed to implementation with (?:the )?(?:\d+ )?(?:already )?approved (?:tasks|plan|changes)(?: \([A-Z0-9–-]+\))?\. Run \/plan-eng-review before (?:the PR is merged|shipping)\.(?: Acceptable if implementation is expected to be fast with CC\.)?$/i.test(q.options[i]!.description?.trim() ?? ''));
if (!runs.some(Boolean) || manual.filter(Boolean).length !== 1 || !labels.every((_, i) => runs[i] || manual[i] || deferred[i])) return null;
return manual.findIndex(Boolean) + 1;
}
/** Every offered explanation must remain a clause about this review handoff. */
function closedNextReviewExplanations(question: string, descriptions: string[]): boolean {
// Validate each complete sentence/line, rather than discarding prose under
// an accepted heading. A new imperative has no navigation subject and
// cannot borrow the preceding sentence's administrative classification.
const navigation = [
/^(?:The )?CEO review (?:is (?:done|complete|cleared)(?: and clears scope and strategy)?|cleared scope and strengthened both test assertions)$/i,
/^The engineering review is the one gate that must pass before shipping \(skip_eng_review is false\)$/i,
/^It checks architecture, code quality, and test design in depth$/i,
/^gstack['’]s shipping gate is the eng review, which checks architecture and test design$/i,
/^it has not run for this plan yet$/i,
/^(?:There is no UI|No UI scope was found), so a design review does not apply$/i,
/^Skipping (?:the )?eng review leaves the (?:required gate unmet, so the readiness dashboard stays NOT CLEARED until someone runs it later|ship dashboard NOT CLEARED)$/i,
/^running it costs a few minutes on a two-test plan$/i,
/^[A-Z] because eng review is the required (?:shipping gate and this plan is now precise enough for it to run quickly|gate and the plan changed since it was written \(two assertions strengthened\), so the tests deserve a second read)$/i,
/^options differ in kind(?: \(which workflow runs next\))?, not coverage [—–-] no completeness score$/i,
/^clear the (?:required )?gate now versus (?:handling reviews on your own schedule|implement first and review later)$/i,
/^Clears the required (?:engineering gate while the plan and its two approved remedies are fresh|shipping gate on the review readiness dashboard)$/i,
/^A second structured pass over the test design catches anything the scope review did not$/i,
/^One more interactive review session before implementation starts$/i,
/^Ends the review chain here$/i,
/^you decide when the eng review runs$/i,
/^No further (?:questions this session|review prompts in this session)$/i,
/^(?:The required eng gate stays unmet and the dashboard remains NOT CLEARED|Dashboard stays NOT CLEARED until an eng review runs)$/i,
/^Second read of the exact assertions and the await-then-count ordering before code is written$/i,
/^A few extra minutes on a plan that is already two tests against existing probes$/i,
/^Move straight to implementing T[1-9]\d* and T[1-9]\d* now$/i,
/^Start the eng review against the updated plan after this review exits$/i,
/^End here$/i,
/^run reviews yourself later$/i,
];
const duration = String.raw`~?\d+(?:\.\d+)?\s*(?:minutes?|mins?|hours?|hrs?|days?|weeks?)`;
const timing = new RegExp(String.raw`\s*\(human:\s*${duration}\s*/\s*CC:\s*${duration}\)$`, 'i');
let metadata = 0;
const body = question.split('\n').slice(1).concat(descriptions.flatMap(text => text.split('\n')));
for (const raw of body) {
const line = raw.trim();
if (!line) continue;
if (/^Project\/branch\/task:/.test(line)) {
// Only the project/mode recap is metadata, never a repair paragraph.
if (++metadata !== 1 || !/^Project\/branch\/task: (?:[\w-]+ on [\w/-]+, \/plan-ceo-review \(HOLD SCOPE\) finished on the payment test-coverage plan|`[\w/-]+`, CEO review of PLAN\.md complete \(HOLD SCOPE, 0 critical gaps, [1-9]\d* assertion fixes approved\))\.$/.test(line)) return false;
continue;
}
if (line === 'Pros / cons:') continue;
if (/^[A-Z][):] /.test(line)) {
const offered = line.replace(/^[A-Z][):] /, '').replace(timing, '').replace(/ \(recommended\)$/, '');
if (!/^(?:Run \/plan-eng-review next|Skip, handle reviews manually)$/.test(offered)) return false;
continue;
}
const prose = line.replace(/^(?:ELI10|Stakes if we pick wrong|Recommendation|Note|Net):\s*/, '')
.replace(/^[✅❌]\s*/, '').replace(timing, '');
const clauses = prose.split(/[.;]\s+|[.]$/).map(s => s.trim()).filter(Boolean);
if (!clauses.length || !clauses.every(clause => navigation.some(pattern => pattern.test(clause)))) return false;
}
return metadata === 1;
}
/** Evidence-only next-review accounting; this never selects a pending option. */
function completedNextReviewBrief(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.sessionId || !call.toolUseId || call.answered !== true || call.failed !== false ||
fp.signature !== `${call.sessionId}:${call.toolUseId}` || call.questions.length !== 1 ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0) ||
!Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length ||
Object.keys(call.answers ?? {}).length !== 1 || !Number.isFinite(Date.parse(call.answeredAt ?? ''))) return false;
const q = call.questions[0]!;
if (q.multiSelect || !/^Next (?:step|review)$/i.test(q.header.trim()) || q.options.length !== 2 ||
fp.options.length !== 2 || !fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label) ||
!q.options.some(o => o.label === call.answers?.[q.question]) || /<gstack-qid/i.test(q.question)) return false;
const question = q.question.trim().replace(/^D\d+\s*[—–-]\s*/i, '');
const lines = question.split('\n').map(line => line.trim()).filter(Boolean);
// Numbered headings and echoed pros/cons are presentation. Require the
// actual navigation query, explicit current CEO closure, and a final brief
// boundary; a new question or directive after that boundary stays work.
if (!/^(?:CEO review (?:is )?(?:complete|done|cleared)\. )?Which review runs next\?$/i.test(lines[0]!) ||
!/^Net:\s+[^\n]+[.!]$/.test(lines.at(-1) ?? '') ||
!/(?:^|[.!?]\s+|^ELI10:\s*)(?:The\s+)?CEO\s+review\s+(?:is\s+)?(?:complete|done|cleared)\b/im.test(question)) return false;
const labels = q.options.map(o => o.label.trim().replace(/^[A-Z][).:]\s*/i, '').replace(/\s*\(recommended\)\s*$/i, ''));
if (labels.filter(label => /^Run \/plan-eng-review next$/i.test(label)).length !== 1 ||
labels.filter(label => /^Skip\s*[,—–-]\s*(?:(?:I['’]ll\s+)?handle reviews manually|manual reviews)$/i.test(label)).length !== 1) return false;
const context = [question, ...q.options.map(o => o.description ?? '')].join('\n')
.replace(/^Stakes if we pick wrong:/m, 'Stakes:');
// Only the next Eng gate can keep the readiness dashboard uncleared.
// Its temporal explanation is not a condition on current CEO closure;
// every other unfinished-work and conditional-closure guard still applies.
const navigationContext = context.replace(
/\b((?:(?:readiness|ship)\s+)?dashboard\s+(?:stays|remains)\s+NOT\s+CLEARED)\s+until\s+(?:someone\s+runs\s+it|(?:an?|the)\s+eng(?:ineering)?\s+review\s+runs)(?:\s+later)?(?=[.!]|\n|$)/gi,
'$1',
);
return q.options.every(o => o.description?.trim()) &&
closedNextReviewExplanations(question, q.options.map(o => o.description ?? '')) &&
!/`{3}|~{3}|(?:^|\n)\s*>|\b(?:example|quoted source)\s*:/im.test(context) &&
!/\bCEO\s+review\b[^.!?\n]{0,80}\b(?:not|never|incomplete|unfinished)\b/i.test(context) &&
/\b(?:Eng|engineering) review\b[^.!?]{0,180}\bgate\b/i.test(context) && closedNavigationContext(navigationContext);
}
/** Classification happens only after one real, successful, fully answered native call. */
export function isCeoCompletionHandoff(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.answered || call.failed || !Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length) return false;
if (completedNextReviewBrief(fp)) return true;
if (manualHandoffIndex(fp) === null) return false;
const q = call.questions[0]!;
// A free-form answer can introduce a new substantive request. Do not
// silently discard it merely because the menu itself was administrative.
return q.options.some(option => option.label === call.answers?.[q.question]);
}
/** Finish this CEO fixture instead of starting another skill; reuse the existing caller-pick hook. */
export function pickCeoCompletionHandoff(
fp: AskUserQuestionFingerprint,
activeCapture: AskUserQuestionFingerprint = fp,
): number | null {
return activeCapture.nativeCall?.answered ? null : manualHandoffIndex(activeCapture);
}
-14
View File
@@ -54,20 +54,6 @@ export function seedCeoPaymentProject(projectDir: string, plan: string): void {
git(['update-ref', 'refs/remotes/origin/main', 'HEAD']);
}
/** Materialized documentation for the revised synthetic DX baseline. The SDK
* implementation is deliberately absent; this does not run or install it. */
export function seedDevexReviewProject(projectDir: string, plan: string): void {
seedPlanReviewProject(projectDir, plan, 'plan-devex-review');
const fixture = path.resolve(import.meta.dir, '../fixtures/devex-existing-sdk');
const files = ['README.md', 'docs/getting-started.md', 'docs/feedback.md', 'docs/reference-v1.md'];
fs.mkdirSync(path.join(projectDir, 'docs'));
for (const file of files) fs.copyFileSync(path.join(fixture, file), path.join(projectDir, file));
const git = (args: string[]) => execFileSync('git', args, { cwd: projectDir, stdio: 'pipe', timeout: 10_000 });
git(['add', ...files]);
git(['-c', 'user.name=Finding fixture', '-c', 'user.email=fixture@gstack.test', 'commit', '-m', 'Seed synthetic SDK documentation']);
git(['update-ref', 'refs/remotes/origin/main', 'HEAD']);
}
export function seedPlanReviewProject(projectDir: string, plan: string, skill: 'plan-ceo-review' | 'plan-eng-review' | 'plan-design-review' | 'plan-devex-review', design?: string): void {
if (!fs.lstatSync(projectDir).isDirectory() || fs.readdirSync(projectDir).length !== 0) {
throw new Error('Plan review fixture requires a fresh private directory');
-19
View File
@@ -1,19 +0,0 @@
import { execFileSync } from 'node:child_process';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { seedCeoFindingProject } from './ceo-finding-fixture';
/** Give the two-test review an existing function to inspect before its first turn. */
export function seedCeoPairedProject(projectDir: string, plan: string): void {
seedCeoFindingProject(projectDir, plan);
const fixture = path.resolve(import.meta.dir, '../fixtures/paired-payment');
fs.mkdirSync(path.join(projectDir, 'src'));
for (const [source, target] of [['README.md', 'README.md'], ['src/payment.ts', 'src/payment.ts'],
['contract.test.ts.fixture', 'contract.test.ts']]) {
fs.copyFileSync(path.join(fixture, source!), path.join(projectDir, target!));
}
const git = (args: string[]) => execFileSync('git', args, { cwd: projectDir, stdio: 'pipe', timeout: 10_000 });
git(['add', 'README.md', 'src/payment.ts', 'contract.test.ts']);
git(['-c', 'user.name=Finding fixture', '-c', 'user.email=fixture@gstack.test', 'commit', '-m', 'Seed existing payment contracts']);
git(['update-ref', 'refs/remotes/origin/main', 'HEAD']);
}
-979
View File
@@ -1,979 +0,0 @@
import { marked } from 'marked';
import { engSetupAUQ, type AskUserQuestionFingerprint } from './claude-pty-runner';
import type { NativePlanQuestionCall } from './plan-count-transcript';
type Seed = 'dispatcher' | 'lookup' | 'email' | 'tests' | 'orders';
type Finding = { seed: Seed; ledgerId: string; phase: string; signature: string };
const plain = (value: string) => value.replace(/[`*_]/g, '').trim();
const option = (value: string) => plain(value).replace(/^[A-D][).]\s*/, '').replace(/\s*\(recommended\)$/i, '');
// A numeric zero and "no" state the same current coverage absence. Keep
// quantified negation and historical/quoted claims out of the seeded defect.
function hasCurrentTestAbsence(value: string): boolean {
const text = prose(value.replace(/"[^"\n]*"|“[^”\n]*”|`[^`\n]*`/g, ''));
return text.split(/(?<=[.!?])\s+|\n/).some(clause => {
if (!current(clause) || /\b(?:previously|formerly|historical|used to|in the past|(?:prior|earlier|old) (?:plan|version))\b/i.test(clause)) return false;
const absent = /\b(?:(?:no|zero|0) (?:automated )?(?:tests?|coverage)|none planned|never (?:runs|executes))\b/gi;
return [...clause.matchAll(absent)].some(match => !/\b(?:not|never|no longer|more than|greater than|less than|at least|at most|over|above|under|below|up to|(?:do|does|did|is|are|was|were|has|have|had|could|would|should|must)n['’]t|can['’]t|won['’]t|cannot)\s+(?:(?:currently|now|yet|still|already|actually|exactly|just|only|have|has|had|contain|contains|include|includes|provide|provides|run|runs|ship|ships)\s+)*$/i.test(clause.slice(0, match.index)));
});
}
// Finite obligations from this fixture's supplied plan. These match the
// behavior under discussion, not decision numbers, option labels, class names
// chosen for a remedy, or a particular generated sentence.
const obligations: Array<{ seed: Seed; subject: RegExp; defect: { test(value: string): boolean }; remedy: RegExp }> = [
{ seed: 'dispatcher', subject: /\b(?:dispatcher|WebhookDispatcher|routing)\b/i,
defect: /\b(?:bypass\w*|skip\w*|separate (?:entry|routing)|second (?:path|front door|routing))\b/i,
remedy: /\b(?:register\w*|reus\w*|route\w*|single routing|one routing)\b/i },
{ seed: 'lookup', subject: /\b(?:SQL|query|lookup|userId|DB|database|parameter)\b/i,
defect: /\b(?:raw|concatenat\w*|interpolat\w*|glue\w*|splice\w*)\b/i,
remedy: /\b(?:bound parameter|bind\w*|parameteriz\w*|prepared statement|ORM|find_by)\b/i },
{ seed: 'email', subject: /\b(?:mail|email|notification|receipt)\b/i,
defect: /\b(?:no error handling|propagat\w*|escape\w*|unhandled|uncaught|rethrow\w*)\b/i,
remedy: /\b(?:rescue|catch|handle|isolate|isolation|enqueue|queue|background job)\b/i },
{ seed: 'tests', subject: /\b(?:tests?|coverage|suite)\b/i,
defect: { test: hasCurrentTestAbsence },
remedy: /\b(?:add|write|implement|handler|unit|integration|regression)\b/i },
{ seed: 'orders', subject: /\b(?:orders?|query|queries)\b/i,
defect: /\b(?:per-order|one query per order|N\+1|(?:fetch\w*|quer\w*)[^.]*loop)\b/i,
remedy: /\b(?:batch\w*|single (?:orders )?query|one (?:bound-parameter )?query|bulk)\b/i },
];
// Use only current prose. Quoted/code blocks never supply a defect, remedy,
// or ledger. Inline code identifiers retain their literal technical names.
function prose(value: string): string {
return marked.lexer(value).filter(t => !['code', 'blockquote', 'html'].includes(t.type))
.map(t => plain(t.raw)).join('\n');
}
function current(value: string): boolean {
return !/^[\x60\"'“‘]/.test(value.trim()) && !/\bno (?:current )?(?:defect|gap|issue|problem)\b/i.test(value) && !/^(?:example|quoted|historical|source|hypothetical|previously|formerly|if|unless)\b/i.test(value.trim()) &&
!/\b(?:this|that|the) (?:finding|issue|decision|defect|assessment|remedy) (?:is|was|has been) (?:already |now )?(?:resolved|fixed|withdrawn|retracted|not current|superseded|historical|quoted)\b/i.test(value);
}
function currentDocumentContext(tokens: ReturnType<typeof marked.lexer>, index: number): boolean {
const headings: Array<{ depth: number; text: string }> = [];
for (const token of tokens.slice(0, index)) if (token.type === 'heading') {
while (headings.length && headings.at(-1)!.depth >= token.depth) headings.pop();
headings.push({ depth: token.depth, text: plain(token.text) });
}
return headings.every(heading => current(heading.text));
}
function currentDocumentSources(tokens: ReturnType<typeof marked.lexer>): string[] {
// A source declaration is metadata, not the historical/source quotation
// excluded by current(). Use one grammar for recognition and currentness,
// including foreign/duplicate declarations; callers still require one PLAN.md.
const label = '(?:Source(?: (?:plan|document|file))?(?: under review)?|(?:Plan|Document|File) under review|(?:Reviewed|Review target|Input) plan)';
const declaration = new RegExp(`^${label}:\\s*`, 'i');
const active = (value: string) => current(value.replace(declaration, 'Review attribution: ')) &&
!/\b(?:history|historical|archiv(?:ed|al)|withdrawn|retracted|superseded|obsolete|cancelled|canceled|not current|no longer current|previously|formerly|hypothetical)\b/i.test(value) &&
!/\b(?:if|unless|might|may|would|could)\b/i.test(value);
const headings: Array<{ depth: number; text: string }> = [];
return tokens.flatMap(token => {
if (token.type === 'heading') {
while (headings.length && headings.at(-1)!.depth >= token.depth) headings.pop();
headings.push({ depth: token.depth, text: plain(token.text) });
}
if (token.type !== 'paragraph' || !headings.every(h => active(h.text)) || /^[`"'“‘]/.test(token.raw.trim())) return [];
const text = plain(token.raw);
if (!active(text)) return [];
return text.split(/(?<=[.!?])\s+|\n/).flatMap(statement => {
const match = declaration.exec(statement.trim());
if (!match || !active(statement)) return [];
const field = statement.trim().slice(match[0].length).trim();
const path = /^([\w./-]+)(?=$|[\s,;!?])/.exec(field);
if (!path) return [field];
// A declaration names one path, optionally followed by source location,
// revision or copy metadata. Unknown tails and additional document paths
// remain non-PLAN records, never a silently discarded second declaration.
const suffix = field.slice(path[1]!.length);
if (suffix.trim() && !/^(?:[.,;]$|\(|@|(?:at|in|on|for)\b|L\d+\b|Validation\b)/i.test(suffix.trim())) return [field];
const references = (value: string) => [...value.matchAll(/\b[\w./-]+\.(?:md|markdown)\b/gi)];
const attribution = suffix.replace(/\(([^()]*)\)/g, (whole, metadata: string) =>
/^(?:copied(?: byte-identically)? (?:in|into|to)|byte-identical to the plan embedded in)\s+/i.test(metadata) &&
references(metadata).length === 1 ? '' : whole);
if (references(attribution).length) return [field];
return [path[1]!.replace(/[.;,]+$/, '')];
});
});
}
// The ledger enumerates "unresolved" while its procedure calls these rows
// pending. Normalize only that unqualified current scalar, never a quoted,
// compound or inactive status. The five existing dispositions keep their rules.
function pendingRowContext(tokens: ReturnType<typeof marked.lexer>, index: number, owner = ''): boolean {
const active = (text: string) => current(text) &&
!/\b(?:historical|archiv(?:ed|al)|withdrawn|retracted|superseded|obsolete|cancelled|canceled|not current|no longer current)\b/i.test(text);
const headings: Array<{ depth: number; text: string }> = [];
for (const token of tokens.slice(0, index + 1)) if (token.type === 'heading') {
while (headings.length && headings.at(-1)!.depth >= token.depth) headings.pop();
headings.push({ depth: token.depth, text: plain(token.text) });
}
return active(owner) && headings.every(heading => active(heading.text));
}
function ledgerStatus(value: string, tokens: ReturnType<typeof marked.lexer>, index: number,
owner: string, evidence: string, sourcePlan: string): string {
const status = plain(value);
if (!/^pending$/i.test(status)) return status;
const sources = currentDocumentSources(tokens);
return pendingRowContext(tokens, index, owner) && current(evidence) &&
sources.length === 1 && sources[0] === 'PLAN.md' && !hasForeignContractSource(evidence, sourcePlan)
? 'unresolved' : '';
}
// An existing suite can provide zero coverage of the new implementation.
// The owned test row supplies that scope; historical quotes and current
// positive/contradictory coverage statements cannot establish its absence.
function excludesCurrentTestTarget(value: string): boolean {
const text = prose(value.replace(/"[^"\n]*"|“[^”\n]*”|`[^`\n]*`/g, ''));
const clauses = text.split(/(?<=[.!?])\s+|\n/).filter(clause => current(clause) &&
!/\b(?:previously|formerly|historical(?:ly)?|used to|(?:prior|earlier|old) (?:plan|version))\b/i.test(clause));
if (clauses.some(clause => /\b(?:not true|false|not the case)\b/i.test(clause) ||
/\b(?:suite|tests?)\s+(?:(?:now|already|also|fully|directly|does|do)\s+)*(?:covers?|exercises?|executes?|tests?|runs?)\s+(?:the\s+)?(?:new|this|current)\s+(?:class|handler|code|path|implementation)\b/i.test(clause))) return false;
return clauses.some(clause => /\b(?:suite|tests?|coverage)\b/i.test(clause) && (
/\b(?:does not|do not|doesn't|don't|never)\s+(?:currently\s+)?(?:cover|exercise|execute|test|run)s?\s+(?:the\s+)?(?:new|this|current)\s+(?:class|handler|code|path|implementation)\b/i.test(clause) ||
/\b(?:suite|tests?)\s+(?:(?:only|still)\s+)?(?:covers?|exercises?|executes?|tests?|runs?)\s+(?:the\s+)?(?:old|prior)\s+(?:class|handler|code|path|implementation)\s*[,;]?\s*(?:but\s+)?not\s+(?:this one|(?:the\s+)?new\s+(?:class|handler|code|path|implementation))\b/i.test(clause)));
}
// A Contracts citation inherits the document's unique current source only
// when its substantive quotation is a complete current source clause. It
// never borrows arbitrary quoted examples or a partial substring elsewhere.
function currentContractQuote(literal: string, sourcePlan: string): boolean {
const normalize = (text: string) => plain(text).replace(/\s+/g, ' ').replace(/[.;]+$/, '').trim();
const active = (text: string) => current(text) && !/\b(?:withdrawn|retracted|superseded|historical|obsolete|no longer current|not current)\b/i.test(text);
const tokens = marked.lexer(sourcePlan);
const clauses = tokens.flatMap((token, index) => token.type === 'paragraph' &&
currentDocumentContext(tokens, index) && active(token.raw)
? plain(token.raw).replace(/\s+/g, ' ').split(/(?<=[.;])\s+/).filter(active).map(normalize) : []);
return active(literal) && clauses.filter(clause => clause === normalize(literal)).length === 1;
}
function hasForeignContractSource(value: string, sourcePlan: string): boolean {
// Only authenticated source quotations can contain incidental code paths.
// Quotation length alone must not hide a conflicting source citation.
const attribution = value.replace(/"([^"\n]+)"|“([^”\n]+)”/g,
(whole, straight, curly) => (straight ?? curly).trim().split(/\s+/).length >= 6 &&
currentContractQuote(straight ?? curly, sourcePlan) ? '' : whole);
const paths = [...attribution.matchAll(/(?:(?:(?:[A-Za-z]:|~)?[\\/]+|\.{1,2}[\\/])(?:[\w.-]+[\\/])*|(?:[\w.-]+[\\/])+)[\w.-]+|\b[\w-]+\.(?:md|markdown)\b/gi)];
return paths.some(match => {
const path = match[0];
if (path === 'PLAN.md') return false;
// A slash alone also joins ordinary prose (read/write, success/failure).
// Filesystem syntax, a filename extension or an explicit reference owns
// a path; a compound in the surrounding explanation does not.
if (!/[\\/]/.test(path) || /^(?:[A-Za-z]:[\\/]|[\\/]|\.{1,2}[\\/]|~[\\/])/.test(path) ||
/\\/.test(path) || /(?:^|[\\/])[^\\/]+\.[\w-]+/.test(path)) return true;
const before = attribution.slice(0, match.index), after = attribution.slice(match.index! + path.length);
return /^(?:[\\/]|:\d+\b|#[\w-]+)/.test(after) ||
(/[`"'“‘<]$/.test(before) && /^[`"'”’>]/.test(after)) ||
/\]\(\s*<?$/.test(before) || /(?:^|\n)\s*\[[^\]]+\]:\s*<?$/.test(before) ||
/\b(?:source(?:\s+(?:plan|file))?|file|path|document|evidence|citation|reference)\s*[:=]\s*$/i.test(before) ||
/\b(?:read|see|consult|from|per|according to|documented in|specified in|cited in)\s+$/i.test(before);
});
}
function currentContractCitation(value: string, sourcePlan: string): boolean {
if (!/^Contracts?:\s*\S/i.test(value) || hasForeignContractSource(value, sourcePlan)) return false;
const normalize = (text: string) => plain(text).replace(/\s+/g, ' ').replace(/[.;]+$/, '').trim();
const active = (text: string) => current(text) && !/\b(?:withdrawn|retracted|superseded|historical|obsolete|no longer current|not current)\b/i.test(text);
const outside = value.replace(/"[^"\n]*"|“[^”\n]*”/g, '');
if (!active(outside)) return false;
const quotes = [...value.matchAll(/"([^"\n]+)"|“([^”\n]+)”/g)].map(match => normalize(match[1] ?? match[2]!))
.filter(literal => literal.split(' ').length >= 6);
if (!quotes.length) return false;
return quotes.every(literal => currentContractQuote(literal, sourcePlan));
}
// A whole quoted ledger value can cite the supplied plan's current prose.
// Authenticate its complete paragraph/sentence, not a substring or a quote
// elsewhere. This does not turn quoted evidence into a seeded defect.
function quotedSourceProposal(value: string, sourcePlan: string): boolean {
const quoted = /^(?:"([^"\n]+)"|'([^'\n]+)'|“([^”\n]+)”|‘([^’\n]+)’)$/u.exec(value.trim());
const literal = quoted?.slice(1).find(part => part !== undefined);
if (!literal || !current(literal)) return false;
const activeSource = (text: string) => current(text) &&
!/\b(?:withdrawn|retracted|superseded|obsolete|historical|archiv(?:ed|al)|(?:no longer|not) current)\b/i.test(text);
const headings: Array<{ depth: number; text: string }> = [];
let matches = 0;
for (const token of marked.lexer(sourcePlan)) {
if (token.type === 'heading') {
while (headings.length && headings.at(-1)!.depth >= token.depth) headings.pop();
headings.push({ depth: token.depth, text: plain(token.text) });
}
if (token.type !== 'paragraph' || !headings.every(h => activeSource(h.text))) continue;
const text = token.raw.trim();
if (!activeSource(text)) continue;
matches += text === literal ? 1 : text.split(/(?<=[.!?])\s+/).filter(sentence => sentence === literal).length;
}
return matches === 1;
}
const mentions = (text: string, id: string) => text.split(/[^A-Za-z0-9_.-]+/).some(token => token.replace(/[.:]$/, '') === id);
function ownedAnswer(fp: AskUserQuestionFingerprint): NativePlanQuestionCall | null {
const call = fp.nativeCall;
if (!call?.sessionId || !call.toolUseId || call.answered !== true || call.failed !== false ||
fp.signature !== `${call.sessionId}:${call.toolUseId}` || call.questions.length !== 1 ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0) ||
!Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length ||
!Number.isFinite(Date.parse(call.answeredAt ?? '')) || Object.keys(call.answers ?? {}).length !== 1) return null;
const q = call.questions[0]!;
if (q.multiSelect || q.options.length < 2 || q.options.length > 4 ||
new Set(q.options.map(o => o.label)).size !== q.options.length ||
!q.options.some(o => o.label === call.answers?.[q.question]) ||
fp.options.length !== q.options.length || !fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label)) return null;
return call;
}
/** Setup may share one native packet. Authenticate the complete answer and
* every offered tab before excluding it; a mixed setup/review packet is not setup. */
function ownedSetupPacket(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.sessionId || !call.toolUseId || call.answered !== true || call.failed !== false ||
fp.signature !== `${call.sessionId}:${call.toolUseId}` || fp.nativeQuestionIndex !== undefined ||
call.questions.length < 2 || call.questions.length > 4 ||
!Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length ||
!Number.isFinite(Date.parse(call.answeredAt ?? '')) ||
Object.keys(call.answers ?? {}).length !== call.questions.length ||
new Set(call.questions.map(q => q.question)).size !== call.questions.length) return false;
const options = call.questions.flatMap(q => q.options.map((o, i) => ({ index: i + 1, label: o.label })));
if (fp.options.length !== options.length || !fp.options.every((o, i) =>
o.index === options[i]!.index && o.label === options[i]!.label)) return false;
if (!call.questions.every(q => !q.multiSelect && q.options.length >= 2 && q.options.length <= 4 &&
new Set(q.options.map(o => o.label)).size === q.options.length &&
typeof call.answers?.[q.question] === 'string' && q.options.some(o => o.label === call.answers[q.question]))) return false;
// These per-question views feed only the bare content classifiers. The
// original packet above owns authentication; a view is never a recorded call.
return call.questions.every(q => setupQuestionContent({
...fp, promptSnippet: `${q.header} ${q.question}`,
options: q.options.map((o, i) => ({ index: i + 1, label: o.label })),
nativeCall: { ...call, questions: [q], answers: { [q.question]: call.answers?.[q.question]! } },
}));
}
/** A question can attribute one offered baseline explicitly "as planned".
* Its title, active ledger proposal and matching native option must agree;
* ELI10 must still assert the current plan's behavior, not quoted history. */
function attributedBaselineDefect(q: NativePlanQuestionCall['questions'][number], proposed: string,
explanation: string, spec: typeof obligations[number]): boolean {
const unquoted = (text: string) => prose(text.replace(/"[^"\n]*"|“[^”\n]*”|(?<!\w)'[^'\n]*'|‘[^’\n]*’|`[^`]*`/g, ''));
const title = unquoted(q.question.split('\n')[0]!).replace(/^D\d+\s*[—–-]\s*/i, '');
if (!current(title) || !title.endsWith('?')) return false;
const alternatives = [...title.matchAll(/(?:^|[:,;]\s*|\bor\s+)([^,;:?]+?)\s+as (?:planned|written)(?=\s*[,;?]|$)/gi)];
if (alternatives.length !== 1) return false;
const baseline = alternatives[0]![1]!.trim();
const normalize = (value: string) => plain(value).toLowerCase().replace(/\s+/g, ' ').trim();
const words = (value: string) => normalize(value).match(/[a-z0-9_]+/g) ?? [];
const expected = words(baseline), actual = words(proposed);
if (!current(baseline) || !spec.subject.test(baseline) || !spec.defect.test(baseline) ||
expected.length < 2 || !actual.some((_, i) => expected.every((word, offset) => actual[i + offset] === word))) return false;
const offered = q.options.filter(o => {
const body=unquoted(`${o.label}\n${o.description ?? ''}`);
return current(body) && !/\b(?:this|that|the) (?:option|alternative|baseline) (?:is|was|has been) (?:already |now )?(?:withdrawn|retracted|rejected|superseded|not current|historical)\b/i.test(body) &&
normalize(option(o.label).replace(/^(?:keep|retain|preserve)\s+/i, '')
.replace(/\s*\(as (?:planned|written)\)\s*$/i, '')) === normalize(baseline);
});
if (offered.length !== 1) return false;
const clauses = unquoted(explanation).split(/(?<=[.!?])\s+|\n/);
return clauses.some(clause => current(clause) && spec.subject.test(clause) &&
/\b(?:the|this) (?:current )?plan\s+\S/i.test(clause) && !spec.remedy.test(clause) &&
!/\b(?:previously|formerly|historical|example|hypothetical|if|unless|not|never|no longer|doesn't|does not)\b/i.test(clause));
}
/** Source requires Current/Proposed/Status/evidence and a cited row ID. It
* does not require heading depth, column order, a Dn(ledger ID) title, or
* native option wording. Pending is valid: the actual ACK precedes the next Edit. */
export function ceoPaymentFinding(fp: AskUserQuestionFingerprint, seedPlan: string, savedPlan: string, trace?: string[]): Finding | null {
const call = ownedAnswer(fp);
if (!call) return null;
const q = call.questions[0]!;
const question = prose(q.question);
const explanation = /^ELI10:\s*(.+)$/m.exec(question)?.[1] ?? question;
if (!question.trim() || !current(question) || !current(explanation)) return null;
const options = q.options.map(o => prose(`${o.label}\n${o.description ?? ''}`)).filter(current);
const tokens = marked.lexer(savedPlan);
const declaredSources = currentDocumentSources(tokens);
const namedSourcePlan = tokens.some(t => t.type === 'paragraph' &&
/(?:^|\n)Source plan:\s*PLAN\.md\b/.test(plain(t.raw)));
const matches: Finding[] = [];
for (const table of tokens.filter(t => t.type === 'table')) {
if (table.type !== 'table') continue;
const column = (meaning: RegExp) => table.header.map((c, i) => meaning.test(plain(c.text)) ? i : -1).filter(i => i >= 0);
const fields = { id: column(/^(?:ID|Decision)\b/i), current: column(/^Current\b/i),
proposed: column(/^Proposed\b/i), status: column(/^Status\b/i), evidence: column(/\b(?:Contract|Evidence)\b/i) };
if (Object.values(fields).some(indices => indices.length !== 1)) continue;
for (const cells of table.rows) {
const read = (key: keyof typeof fields) => plain(cells[fields[key][0]!]!.text);
const owner = read('id'), id = owner.split(/\s/, 1)[0]!.replace(/[.:]$/, '');
const pending = /^pending$/i.test(read('status'));
const status = ledgerStatus(cells[fields.status[0]!]!.text, tokens, tokens.indexOf(table), owner, read('evidence'), seedPlan);
const sourceBound = /\bPLAN\.md\b/.test(read('evidence')) ||
(namedSourcePlan && /\bEvidence:\s*plan text\b/i.test(read('evidence')));
if (!id || !mentions(question, id) || !/^(?:unresolved|approved|reopened|deferred|declined)\b/i.test(status) || !sourceBound) continue;
// A row can contain its proposals directly or cite a separate saved
// comparison bearing the same ID. Heading spelling/depth is immaterial.
const blocks = tokens.map((t, i) => t.type === 'heading' && mentions(plain(t.text), id) ? i : -1).filter(i => i >= 0);
if (pending && blocks.filter(index => pendingRowContext(tokens, index)).length > 1) continue;
const proposals: Array<{ body: string; phase: string; active: boolean }> = [{ body: read('proposed'), phase: 'ledger row', active: currentDocumentContext(tokens, tokens.indexOf(table)) }];
for (const start of blocks) {
const heading = tokens[start]!;
if (heading.type !== 'heading') continue;
let end = start + 1;
while (end < tokens.length && !(tokens[end]!.type === 'heading' && (tokens[end] as any).depth <= heading.depth)) end++;
const preceding = tokens.slice(0, start).filter(t => t.type === 'heading' && t.depth < heading.depth).at(-1);
proposals.push({ body: prose(tokens.slice(start + 1, end).map(t => t.raw).join('')), phase: preceding?.type === 'heading' ? preceding.text : heading.text, active: current(plain(heading.text)) && currentDocumentContext(tokens, start) && (!pending || pendingRowContext(tokens, start)) });
}
for (const spec of obligations) {
const row = `${owner} ${read('evidence')} ${read('current')}`;
// Current holds existing/approved behavior. A correct baseline can
// still have a defective pending alternative in Proposed; keep that
// defect bound to this active row, not a copied comparison elsewhere.
const defectValue = (field: 'current' | 'proposed') => spec.seed === 'tests'
? cells[fields[field][0]!]!.text : read(field);
const defectExplanation = spec.seed === 'tests'
? /^ELI10:\s*(.+)$/m.exec(q.question)?.[1] ?? q.question : explanation;
const scopedTestAbsence = spec.seed === 'tests' && declaredSources.length === 1 && declaredSources[0] === 'PLAN.md' &&
currentDocumentContext(tokens, tokens.indexOf(table)) && /\btests?\b/i.test(owner) &&
/^(?:unresolved|reopened)\b/i.test(status) && current(read('current')) &&
/^(?:None|zero|0|no (?:new )?(?:automated )?tests?)\.?$/i.test(cells[fields.proposed[0]!]!.text.trim()) &&
excludesCurrentTestTarget(cells[fields.current[0]!]!.text) && excludesCurrentTestTarget(defectExplanation);
const pendingDefect = /^(?:unresolved|reopened)\b/i.test(status) &&
current(read('proposed')) && spec.subject.test(read('proposed')) && spec.defect.test(defectValue('proposed'));
const checks = {
seedPlan: spec.subject.test(seedPlan) && spec.defect.test(seedPlan),
rowSubject: spec.subject.test(row),
rowDefect: spec.defect.test(defectValue('current')) || pendingDefect || scopedTestAbsence,
questionSubject: spec.subject.test(question),
explanationDefect: spec.defect.test(defectExplanation) || scopedTestAbsence ||
(pendingDefect && declaredSources.length <= 1 && declaredSources.every(source => source === 'PLAN.md') &&
currentDocumentContext(tokens, tokens.indexOf(table)) && attributedBaselineDefect(q, read('proposed'), explanation, spec)),
operativeOption: options.some(o => spec.remedy.test(o) && spec.subject.test(o)),
proposal: proposals.some(p => (!(scopedTestAbsence || pending) || p.active) && current(p.body) && spec.remedy.test(p.body) && spec.subject.test(p.body)),
};
if (trace && (checks.rowSubject || checks.questionSubject))
trace.push(`${spec.seed}@${id}: ${Object.entries(checks).map(([name, ok]) => `${name}=${ok ? 'yes' : 'no'}`).join(' ')}`);
if (!checks.seedPlan || !checks.rowSubject || !checks.rowDefect || !checks.questionSubject || !checks.explanationDefect) continue;
const proposal = proposals.find(p => (!(scopedTestAbsence || pending) || p.active) && current(p.body) && spec.remedy.test(p.body) && spec.subject.test(p.body));
if (checks.operativeOption && proposal) matches.push({ seed: spec.seed, ledgerId: id, phase: proposal.phase, signature: fp.signature });
}
}
}
return matches.length === 1 ? matches[0]! : null;
}
function setupQuestion(fp: AskUserQuestionFingerprint): boolean {
const call = ownedAnswer(fp);
if (!call) return false;
return setupQuestionContent(fp);
}
function setupQuestionContent(fp: AskUserQuestionFingerprint): boolean {
const q = fp.nativeCall!.questions[0]!;
const title = prose(q.question).split('\n')[0]!;
const labels = q.options.map(o => option(o.label));
if (/\b(?:skill routing|routing rules)\b/i.test(title) && /\bCLAUDE\.md\b/i.test(title))
return labels.length === 2 && labels.some(l => /\b(?:add|enable|include|append)\b.*\brouting\b/i.test(l)) && labels.some(l => /\b(?:no thanks|skip|manually|manual)\b/i.test(l));
// Authentication belongs to the complete original packet or single-call
// wrapper; setup content needs no working-plan file yet.
if (prose(q.question).trim() && current(prose(q.question)) && engSetupAUQ(fp)) return true;
// Preserve the existing label-wrapper contract; the shared predicate
// expects unnumbered action labels while this older route accepts wrappers.
if (/\bcross[- ]project learnings\b/i.test(title) && /\b(?:enable|search)\b/i.test(title))
return labels.length === 2 && labels.some(l => /\benable\b.*\bcross[- ]project\b/i.test(l)) && labels.some(l => /\bproject[- ]scoped\b/i.test(l));
const modes = labels.map(l => l.match(/\b(?:SCOPE EXPANSION|SELECTIVE EXPANSION|HOLD SCOPE|SCOPE REDUCTION)\b/g));
if (labels.length === 4 && modes.every(found => found?.length === 1) && new Set(modes.flat()).size === 4) return true;
if (/\b(?:scope|review target)\b/i.test(title) && labels.some(l => /skip\s+interview|plan\s+immediately/i.test(l))) return true;
if (/\boffice-hours\b/i.test(title) && labels.length === 2 && labels.some(l => /\brun\b.*office-hours/i.test(l)) && labels.some(l => /^skip\b/i.test(l))) return true;
const remedyEvidence = obligations.some(spec => spec.subject.test(q.question) && spec.defect.test(q.question) &&
q.options.some(o => spec.remedy.test(`${o.label} ${o.description ?? ''}`)));
return !remedyEvidence && /\b(?:which|choose|select)\b.*\bapproach\b/i.test(title) && /^Approach$/i.test(q.header);
}
function todoDecision(fp: AskUserQuestionFingerprint): boolean {
const q = fp.nativeCall!.questions[0]!;
return /\bTODO(?:S\.md|s|[- ]\d+)?\b/i.test(q.header + ' ' + q.question.split('\n')[0]) &&
q.options.some(o => /^(?:add|build|implement|remove|defer|skip)\b/i.test(option(o.label)));
}
/** Count other real choices by their saved decision identity, not a defect
* vocabulary. A row alone is insufficient: its own complete comparison must
* bind every offered native option. This grants count credit, not approval. */
function recordedDecision(fp: AskUserQuestionFingerprint, savedPlan: string, sourcePlan: string): { ledgerId: string; phase: string } | null {
const call = ownedAnswer(fp);
if (!call) return null;
const q = call.questions[0]!, question = prose(q.question);
if (!question.trim() || !current(question)) return null;
const title = question.split('\n')[0]!;
const tokens = marked.lexer(savedPlan);
// A current document may declare its source once and cite that plan's
// sections in each row. An unrelated mention elsewhere is not provenance.
const currentContext = (index: number) => sectionContext(tokens, index) &&
(tokens[index]?.type !== 'heading' || activeSection(plain(tokens[index].text)));
const sourceRecords = currentDocumentSources(tokens);
const namedSource = sourceRecords.length === 1 && sourceRecords[0] === 'PLAN.md';
const lineCitation = (evidence: string) => {
const cited = /^Plan lines?\s+([1-9]\d*(?:\s*[-–—]\s*[1-9]\d*)?(?:\s*,\s*[1-9]\d*(?:\s*[-–—]\s*[1-9]\d*)?)*)\s*:/i.exec(evidence);
return Boolean(cited && cited[1]!.split(',').every(range => {
const bounds=range.trim().split(/\s*[-–—]\s*/).map(Number), first=bounds[0]!, last=bounds.at(-1)!;
return Number.isSafeInteger(first) && Number.isSafeInteger(last) && first<=last && last<=sourcePlan.split('\n').length;
}));
};
const inheritedSource = (evidence: string) => namedSource &&
(/\bEvidence:\s*plan text\b|\bplan\s+§\s*\S|\bplan\s+sections?\s+\S|^Plan(?: contract)?:\s*\S/i.test(evidence) ||
lineCitation(evidence) || currentContractCitation(evidence, sourcePlan));
// A section citation can name the source in its current heading instead
// of a special document-wide declaration. Resolve every cited section
// against the actual input, and require an attributed current section.
const activeSection = (text: string) => current(text) &&
// The prescribed answered-decision history is separate from a reopened
// row's current payload; it cannot supply or duplicate that comparison.
!/^Answered decisions?\b/i.test(text) &&
!/\b(?:historical|archiv(?:ed|al)|withdrawn|retracted|superseded|obsolete|not current|no longer current)\b/i.test(text);
const sectionContext = (document: ReturnType<typeof marked.lexer>, index: number) => {
const headings: Array<{ depth: number; text: string }> = [];
// Enter the current heading before checking context: a completed sibling
// (and its descendants) is not an ancestor of the section that follows.
for (const token of document.slice(0, index + 1)) if (token.type === 'heading') {
while (headings.length && headings.at(-1)!.depth >= token.depth) headings.pop();
headings.push({ depth: token.depth, text: plain(token.text) });
}
return headings.every(heading => activeSection(heading.text));
};
const sectionCitation = (raw: string) => {
const evidence = plain(raw.replace(/`[^`]*`|"[^"\n]*"|“[^”\n]*”|(?<!\w)'[^'\n]*'|‘[^’\n]*’/g, ''));
if (!activeSection(evidence) || hasForeignContractSource(raw, sourcePlan) || sourceRecords.length > 1 || sourceRecords.some(source => source !== 'PLAN.md')) return false;
const attributed = tokens.flatMap((token, index) => {
if (token.type !== 'heading' || !sectionContext(tokens, index) || !activeSection(plain(token.text))) return [];
const match = /^(.+?)(?:\s+retained)?\s+\(from\s+([^()]+)\)$/i.exec(plain(token.text));
return match ? [{ section: match[1]!.toLowerCase(), source: match[2]! }] : [];
});
if (!attributed.length || attributed.some(row => row.source !== 'PLAN.md') ||
new Set(attributed.map(row => row.section)).size !== attributed.length) return false;
const sourceTokens = marked.lexer(sourcePlan);
const headings = sourceTokens.flatMap((token, index) => token.type === 'heading' &&
sectionContext(sourceTokens, index) && activeSection(plain(token.text)) ? [plain(token.text).replace(/\s+retained$/i, '').toLowerCase()] : []);
const references = [...evidence.matchAll(/§\s*/g)].map(match => {
const tail = evidence.slice(match.index! + match[0].length).toLowerCase();
return headings.filter(heading => tail.startsWith(heading) && /^(?:\s|[.,;:]|$)/.test(tail.slice(heading.length)));
});
return references.length > 0 && references.every(matches => matches.length === 1) &&
references.some(matches => attributed.some(row => row.section === matches[0]));
};
// The same option may give both dimensions as a parenthesized tuple,
// with the value before or after its field, or a bare finite effort size.
// Risk must remain explicit. Inventory every metadata tuple before accepting
// one so mixed compact/full forms cannot hide duplicate or invalid claims.
const optionFacts = (raw: string) => {
const visible = raw.replace(/`+[^`]*`+|"[^"\n]*"|“[^”\n]*”|(?<![\p{L}\p{N}])'[^'\n]*'(?![\p{L}\p{N}])|‘[^’\n]*’/gu,
match => ' '.repeat(match.length));
const firstTradeoff = visible.search(/\b(?:Pros|Cons)\s*:/i);
const claims = [...visible.matchAll(/\(([^()]+)\)/g)].filter(match =>
(firstTradeoff < 0 || match.index! < firstTradeoff) && /\brisk\b/i.test(match[1]!) &&
(/\beffort\b/i.test(match[1]!) || /[,;]/.test(match[1]!)));
if (!claims.length) return raw;
if (claims.length !== 1) return null;
const match = claims[0]!, before = visible.slice(0, match.index).trimEnd();
const after = visible.slice(match.index! + match[0].length);
if (/\b(?:not|never|no longer|previously|formerly|historical(?:ly)?|hypothetical(?:ly)?|quoted|withdrawn|retracted)(?:[\s,:;.—–-]+(?:currently|now|actually|exactly|only|still|just|estimated?|rated?|rating|as|at|effort|risk|tuple|metadata))*[\s,:;.—–-]*$/i.test(before) ||
/\b(?:this|that|the) (?:estimate|tuple|rating|metadata|effort|risk) (?:is|was|has been) (?:already |now )?(?:withdrawn|retracted|not current|no longer (?:current|valid)|superseded|historical|quoted)\b/i.test(visible) ||
!/^(?:\s*[.,;]|\s*$)/.test(after)) return null;
const fields = match[1]!.split(/\s*[,;]\s*/).map(part => {
const forward = /^(effort|risk)\s*:?\s+(.+)$/i.exec(part.trim());
const reverse = /^(.+?)\s+(effort|risk)$/i.exec(part.trim());
return forward ? [forward[1]!.toLowerCase(), forward[2]!] : reverse ? [reverse[2]!.toLowerCase(), reverse[1]!]
: /^(?:S|M|L|XL)$/i.test(part.trim()) ? ['effort', part.trim()] : [];
});
const facts = Object.fromEntries(fields.filter(field => field.length === 2));
const risk = /^(low|medium|high)(?:\s*(?:[-–—]|\bto\b)\s*(low|medium|high))?$/i.exec(facts.risk ?? '');
const levels = ['low','medium','high'];
if (fields.length !== 2 || Object.keys(facts).length !== 2 ||
!/^(?:S|M|L|XL)$/i.test(facts.effort ?? '') || !risk ||
(risk[2] && levels.indexOf(risk[1]!.toLowerCase()) >= levels.indexOf(risk[2]!.toLowerCase()))) return null;
return raw.slice(0, match.index) + `. Effort ${facts.effort}. Risk ${facts.risk}.` + raw.slice(match.index! + match[0].length);
};
const optionText = (raw:string) => raw
.replace(/((?:this|that|the) (?:option|alternative|baseline) (?:is|was|has been)\s+(?:(?:already|now)\s+)?)["“'‘]([^"”'’\n]+)["”'’]/gi,'$1$2')
.replace(/"[^"\n]*"|“[^”\n]*”|(?<!\w)'[^'\n]*'|‘[^’\n]*’/g,'');
const withdrawnOption = /\b(?:this|that|the) (?:option|alternative|baseline) (?:is|was|has been) (?:already |now )?(?:withdrawn|retracted|rejected|superseded|not current|no longer (?:current|valid)|historical|quoted)\b/i;
// The skill requires complete per-option facts, not a GFM option table.
// Code and quoted children cannot supply a prose/list option's fields.
const proseOption = (parts: readonly any[]) => {
const paragraphs = parts.filter(part => part.type === 'paragraph' || part.type === 'text');
const first = paragraphs[0];
if (!first) return null;
const normalized = optionFacts(paragraphs.map(part => part.raw).join('\n'));
if (normalized === null) return null;
const text = plain(normalized);
const label = first.tokens?.[0]?.type === 'strong' ? plain(first.tokens[0].text)
: /^([A-D][).:]\s+.+?)\s+[—–-]\s+/i.exec(text)?.[1]
?? /^([A-D][).:]\s+.+?)[.:]\s+/i.exec(text)?.[1]
// A plain label can own the next line's full option facts. A
// single-line fragment cannot borrow fields from another paragraph.
?? (first.type === 'paragraph' && tokens.indexOf(first) >= 0 &&
first.raw.trim().includes('\n') && currentContext(tokens.indexOf(first)) &&
/^[A-D][).:]\s+\S/i.test(plain(first.raw.split('\n')[0]!))
? plain(first.raw.split('\n')[0]!) : undefined);
if (!label || !/^[A-D][).:]\s+\S/i.test(label)) return null;
const details = text.slice(label.length).replace(/^[.:\s—–-]+/, '');
// Mask quoted/code field names without changing offsets. A real field
// may follow a quoted sentence, but the quotation cannot supply a field.
const fieldText = plain(normalized.replace(/`[^`]*`|"(?:\\.|[^"\\])*"|“[^”]*”|(?<![\p{L}\p{N}])'[^']*'(?![\p{L}\p{N}])|‘[^’]*’/gu, raw => {
const literal = raw.replace(/[`*_]/g, '');
const ending = /[.!?,;]["”'’]$/.exec(literal)?.[0] ?? '';
return literal.slice(0, literal.length - ending.length).replace(/[^\s]/g, ' ') + ending;
})).slice(text.length - details.length);
const facts = [...fieldText.matchAll(/(?:^|[.!?,;]["”'’]?\s+|\n\s*)(Effort(?: estimate)?|Risk(?: level)?|Pros|Cons)\s*:?\s+/gi)];
const fields = Object.fromEntries(facts.map((fact, index) => [fact[1]!.split(' ')[0]!.toLowerCase(),
details.slice(fact.index! + fact[0].length, facts[index + 1]?.index ?? details.length).trim()]));
// A Cons condition states a contingent cost of this current alternative;
// it does not make the option, its promised benefit or its metadata
// hypothetical. Keep withdrawal/source/history checks on the clause and
// the whole option, and keep every other field's currentness unchanged.
const currentFact = (field: string, value: string) => current(field === 'cons'
? value.replace(/^(?:if|unless)\s+(?=\S)/i, '') : value);
const complete = facts.length === 4 && Object.keys(fields).length === 4 && current(text) && !withdrawnOption.test(optionText(text)) &&
['effort', 'risk', 'pros', 'cons'].every(field => fields[field] && currentFact(field, fields[field]!)) &&
/^(?:S|M|L|XL)\b/i.test(fields.effort!) && /^(?:low|medium|high)\b/i.test(fields.risk!);
return { label, summary: text, bindingText: label + ' ' + details.slice(0, facts[0]?.index ?? details.length), complete };
};
// The checkpoint also saves the native Question/Header and unchanged full
// option descriptions. These use native ✅/❌ tradeoffs, not prose-fallback
// field names. Match the whole current record without borrowing old tables.
const exactNativeFields = (section: ReturnType<typeof marked.lexer>) => {
const line = (value: string) => value.trim()
.replace(/^\*\*(Question|Header):\*\*\s*/, '$1: ')
.replace(/^\*\*(Question|Header)\*\*:\s*/, '$1: ')
.replace(/^\*\*([A-D][).:]\s+.+)\*\*$/, '$1');
const lines = (value: string) => value.replace(/\r\n/g, '\n').split('\n').map(line).filter(Boolean);
const saved = section.filter(token => token.type === 'paragraph' && currentContext(tokens.indexOf(token))).flatMap(token => lines(token.raw));
const questions = saved.flatMap((value, i) => /^Question:/.test(value) ? [i] : []);
const headers = saved.flatMap((value, i) => /^Header:/.test(value) ? [i] : []);
if (questions.length !== 1 || headers.length !== 1 || !q.header.trim()) return false;
const prefixes = q.options.flatMap(offered => /^([A-D])[).:]\s+/.exec(offered.label)?.[1] ?? []);
if (new Set(prefixes).size !== prefixes.length) return false;
const assigned = new Set(prefixes);
const options = q.options.map((offered, index) => {
const prefix = /^([A-D])[).:]\s+/.exec(offered.label);
if (prefix && /^[A-D][).:]\s+/.test(offered.label.slice(prefix[0].length))) return null;
const id = prefix?.[1] ?? ['A', 'B', 'C', 'D'].find(value => !assigned.has(value));
if (!id) return null;
assigned.add(id);
const description = prose(offered.description ?? '');
const tradeoffs = [...description.matchAll(/([✅❌])\s*([^✅❌]+)/g)];
if (!description.trim() || !current(description) || withdrawnOption.test(optionText(description)) ||
!/\bEffort(?: estimate)?\s*:?\s+(?:S|M|L|XL)\b/i.test(description) ||
!/\bRisk(?: level)?\s*:?\s+(?:low|medium|high)\b/i.test(description) ||
tradeoffs.filter(part => part[1] === '✅').length < 2 || !tradeoffs.some(part => part[1] === '❌') ||
tradeoffs.some(part => !/[A-Za-z0-9]/.test(part[2]!))) return null;
return `${prefix ? offered.label : `${id}) ${offered.label}`}\n${offered.description}`;
});
if (options.some(value => value === null)) return false;
const fields = [`Question: ${q.question}`, `Header: ${q.header}`];
if (headers[0]! < questions[0]!) fields.reverse();
const expected = lines([...fields, ...options].join('\n'));
const start = Math.min(questions[0]!, headers[0]!);
if (!saved.slice(0, start).every(activeSection)) return false;
const actual = saved.slice(start);
return actual.length === expected.length && actual.every((value, index) => value === expected[index]);
};
const selector = (label: string) => /^([A-D])[.):]\s*/i.exec(plain(label))?.[1]?.toUpperCase();
const labelWords = (label: string) => (option(label).toLowerCase().match(/[a-z][a-z0-9_]{3,}/g) ?? [])
.filter(word => !['recommended', 'option', 'only', 'plan', 'planned', 'written', 'keep', 'same', 'full'].includes(word));
// Terminal punctuation and a status suffix are presentation, not a choice.
const caption = (value: string) => option(value).replace(/^[A-D]:\s*/i, '').replace(/\s*\((?:plan )?as (?:written|planned)\)\.?$/i, '').replace(/[.:]$/, '').trim();
const words = (value: string) => (caption(value).toLowerCase().match(/[a-z0-9_]+/g) ?? [])
.map(word => word === 'via' ? 'through' : word);
const completeCaption = (offered: string, saved: string, summary: string) => {
const a = selector(offered), b = selector(saved);
if (a && a !== b) return false;
const left = words(offered), right = words(saved + ' ' + summary);
// Abbreviations may omit detail, but an unlettered saved caption cannot
// add an action or narrow its scope. A terminal 'in place' is presentation.
const savedCaption = words(caption(saved).replace(/ in place$/i, ''));
if (!a && savedCaption.some(word => !left.includes(word))) return false;
// A lettered grid may abbreviate a terminal "only" qualifier; never
// discard an action's internal scope or a negation while binding it.
if (a && left.at(-1) === 'only' && !right.includes('only')) left.pop();
if (['no', 'not', 'never', 'without'].some(word => left.includes(word) !== right.includes(word))) return false;
if (!left.length || (!a && left.length < 2) || left[0] !== right[0]) return false;
let cursor = 0;
return left.every(word => { const index = right.indexOf(word, cursor); cursor = index + 1; return index >= 0; });
};
const sameOption = (offered: string, saved: string, summary: string) => caption(offered).toLowerCase() === caption(saved).toLowerCase() ||
Boolean(selector(offered) && selector(offered) === selector(saved) &&
labelWords(offered).some(word => labelWords(saved + ' ' + summary).includes(word))) ||
(!selector(offered) && completeCaption(offered, saved, summary));
// Bare grid columns supply no action text. Bind their declaration to the
// whole native caption, preserving targets, scope, counts and negation.
// Assertion summaries may omit "assert full", count precision, a mock
// already named in the native brief, and source-bound scalar call arguments.
const declaredOption = (offered: typeof q.options[number], saved: string) => {
const assignment = '[a-z_][a-z0-9_]*=(?:[0-9]+|[a-z_][a-z0-9_]*)';
const argumentsKey = (value: string) => {
const pairs = value.match(new RegExp(assignment, 'gi')) ?? [];
return pairs.length && new Set(pairs.map(pair => pair.split('=')[0])).size === pairs.length
? pairs.sort().join(',') : null;
};
const sourceArguments = (value: string) => {
const key = argumentsKey(value), sourceTokens = marked.lexer(sourcePlan);
if (!key) return false;
return sourceTokens.some((token, index) => {
if (!sectionContext(sourceTokens, index)) return false;
const parts = token.type === 'paragraph' ? [token] : token.type === 'list'
? token.items.flatMap(item => item.tokens.filter(part => part.type === 'text' || part.type === 'paragraph')) : [];
return parts.some(part => {
const text = prose(part.raw.replace(/"[^"\n]*"|“[^”\n]*”|(?<!\w)'[^'\n]*'|‘[^’\n]*’/g, '')).replace(/\s+/g, ' ');
if (!activeSection(text) || /\b(?:not|never|if|unless|hypothetical|previously|formerly)\b/i.test(text)) return false;
const calls = new RegExp(`\\bcall\\s+[a-z_][a-z0-9_.]*(?:\\(\\))?\\s+with\\s+(${assignment}(?:\\s*(?:,|\\band\\b)\\s*${assignment})*)(?=\\s*(?:[,.;]|$))`, 'gi');
return [...text.matchAll(calls)].some(call => argumentsKey(call[1]!) === key);
});
});
};
const normalize = (value: string) => {
let text = caption(value);
if (/^assert\s+/i.test(text)) {
text = text.replace(/^assert\s+(?:full\s+)?/i, '')
.replace(/\bexactly\s+(?=(?:[0-9]+|one|two|three|four)\b)/gi, '');
if (/\bmock\b/i.test(prose(offered.description ?? '')))
text = text.replace(/\bmock\s+(?=[a-z_][a-z0-9_]*\s+call\b)/gi, '');
text = text.replace(new RegExp(`(\\bcall)\\s+with\\s+(${assignment}(?:,\\s*${assignment})*)$`, 'i'),
(whole, call, args) => sourceArguments(args) ? call : whole);
}
return text.toLowerCase().replace(/\+|\bplus\b/g, ' and ').match(/[a-z0-9_]+|[^\s.,]/g) ?? [];
};
const left = normalize(offered.label), right = normalize(saved);
return selector(offered.label) === selector(saved) && left.length > 0 &&
left.length === right.length && left.every((word, index) => word === right[index]);
};
// A saved "as planned" alternative names the owned baseline. Resolve that
// reference before ordinary caption matching; a letter or a shared word is
// insufficient, and retaining a baseline cannot silently append an action.
const baselineCaption = (value: string) => option(value).replace(/^[A-D]:\s*/i, '').replace(/[.]$/, '').trim();
const baselineWords = (value: string) => baselineCaption(value).toLowerCase().match(/[a-z0-9_]+/g) ?? [];
const sameBaseline = (a: string, b: string) => baselineCaption(a).replace(/\s+/g, ' ').toLowerCase() ===
baselineCaption(b).replace(/\s+/g, ' ').toLowerCase();
const retainedCaption = (value: string) => baselineCaption(value)
.replace(/^(?:keep|retain|preserve)\s+/i, '')
.replace(/^as (?:planned|written):\s*/i, '')
.replace(/\s*\((?:plan )?as (?:planned|written)\)$/i, '');
const savedBaseline = (saved: { label: string; bindingText: string }) => {
const label = baselineCaption(saved.label);
const tail = saved.bindingText.slice(saved.label.length).trim();
if (/^as (?:planned|written)\b/i.test(label)) {
const caption = retainedCaption(label.replace(/^as (?:planned|written):?\s*/i, ''));
return { generic: !caption, caption };
}
const suffix = /^(.*?)\s*(?:\((?:plan )?as (?:planned|written)\)|,\s*as (?:planned|written))$/i.exec(label);
if (suffix) return { generic: false, caption: retainedCaption(suffix[1]!) };
if (/^\((?:plan )?as (?:planned|written)\)(?:\s|[—–-]|$)/i.test(tail)) return { generic: false, caption: retainedCaption(label) };
return null;
};
const matches: Array<{ ledgerId: string; phase: string }> = [];
for (const table of tokens.filter(t => t.type === 'table')) {
if (table.type !== 'table') continue;
const index = (meaning: RegExp) => table.header.flatMap((cell, i) => meaning.test(plain(cell.text)) ? [i] : []);
const fields = { id: index(/^(?:ID|Decision)\b/i), evidence: index(/\b(?:Contract|Evidence)\b/i),
current: index(/^Current\b/i), proposed: index(/^Proposed\b/i), status: index(/^Status\b/i) };
if (Object.values(fields).some(found => found.length !== 1)) continue;
for (const cells of table.rows) {
const read = (key: keyof typeof fields) => plain(cells[fields[key][0]!]!.text);
const id = read('id').split(/\s/, 1)[0]!.replace(/[.:]$/, '');
const status = ledgerStatus(cells[fields.status[0]!]!.text, tokens, tokens.indexOf(table), read('id'), read('evidence'), sourcePlan);
const quotedProposal = !current(read('proposed')) && /^(?:unresolved|reopened)\b/i.test(status) &&
(!sourceRecords.length || namedSource) && currentContext(tokens.indexOf(table)) &&
quotedSourceProposal(cells[fields.proposed[0]!]!.text, sourcePlan);
const sectionEvidence = sectionCitation(cells[fields.evidence[0]!]!.text);
const headingCitation = !namedSource && /§/.test(read('evidence')) && tokens.some(token =>
token.type === 'heading' && /\(from\s+[^()]+\)$/i.test(plain(token.text)));
if (headingCitation && !sectionEvidence) continue;
if (!id || !mentions(title, id) || !/^(?:unresolved|reopened|approved|deferred|declined)\b/i.test(status) ||
!read('current') || !read('proposed') || read('current') === read('proposed') ||
!current(read('evidence')) || (!current(read('proposed')) && !quotedProposal)) continue;
if (!/\bPLAN\.md\b/.test(read('evidence')) && !inheritedSource(read('evidence')) && !sectionEvidence) continue;
const contractCitation = /^Contracts?:/i.test(read('evidence'));
if (contractCitation && (!currentContext(tokens.indexOf(table)) || hasForeignContractSource(read('evidence'), sourcePlan))) continue;
if (sectionEvidence && !sectionContext(tokens, tokens.indexOf(table))) continue;
// A named current record is also valid as a plain/bold paragraph.
// A bare Row marker inherits only its enclosing currentDecision heading;
// incidental row mentions and quoted/code tokens cannot own a comparison.
const paragraphRecord = (index: number) => {
const token = tokens[index];
if (token?.type !== 'paragraph' || !currentContext(index) || !current(plain(token.raw))) return false;
const marker = plain(token.raw).split('\n')[0]!;
const escaped = id.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
const parentIndex = tokens.slice(0, index).findLastIndex(t => t.type === 'heading');
const parent = tokens[parentIndex];
// An immediate same-row marker describes its named heading's record.
// A marker after any record content remains a separate declaration.
if (parent?.type === 'heading' && /^currentDecision\b/i.test(plain(parent.text)) &&
mentions(plain(parent.text), id) && tokens.slice(parentIndex + 1, index).every(t => t.type === 'space')) return false;
if (new RegExp(`^currentDecision\\s*(?:[:(—–-]\\s*)?${escaped}(?=$|[\\s):—–-])`, 'i').test(marker)) return activeSection(marker);
return parent?.type === 'heading' && /^currentDecision\b/i.test(plain(parent.text)) &&
new RegExp(`^Row\\s+${escaped}(?=$|[\\s:—–-])`, 'i').test(marker) && activeSection(marker);
};
const anchors = tokens.flatMap((t, i) =>
(t.type === 'heading' && current(plain(t.text)) && mentions(plain(t.text), id)) ||
(t.type === 'paragraph' && /^(?:Options|Approaches|Comparison)\b/i.test(plain(t.raw)) && mentions(plain(t.raw), id)) ||
paragraphRecord(i) ? [i] : []);
// A row reference in a coverage/task heading does not declare another
// saved decision. Keep broad legacy discovery, but count ownership only
// where a record is declared or its own fields/comparison begin. Explicit
// empty/incomplete records still conflict; never borrow a child record.
const recordAnchors = anchors.filter(start => {
const anchor = tokens[start]!;
const heading = plain(anchor.raw).replace(/^#+\s*/, '');
const rowName = id.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
const kind = '(?:decision|review|options|approaches|comparison)';
const declared = new RegExp(`^(?:(?:current|pending)\\s+)?(?:${kind}\\s+(?:for\\s+)?${rowName}\\b|${rowName}\\s+${kind}\\b)`, 'i');
if (paragraphRecord(start) || /^currentDecision\b/i.test(heading) || declared.test(heading) ||
(anchor.type === 'paragraph' && /^(?:Options|Approaches|Comparison)\b/i.test(plain(anchor.raw)))) return true;
let end = start + 1;
while (end < tokens.length && tokens[end]!.type !== 'heading') end++;
return tokens.slice(start + 1, end).some(token => {
if (token.type === 'paragraph') return !/^[`"'“‘]/.test(token.raw.trim()) &&
/^(?:(?:Question|Header):|[A-D][).:]\s+\S)/m.test(plain(token.raw));
if (token.type === 'list') return token.items.some(item => !/^[`"'“‘]/.test(item.text.trim()) &&
/^[A-D][).:]\s+\S/.test(plain(item.text)));
const columns = token.type === 'table' ? token.header.map(cell => plain(cell.text)) :
token.type === 'code' ? (token.text.split('\n').find(line => line.includes('|')) ?? '').split('|').map(plain) : [];
return columns.some(column => /^(?:Option|Approach)\b/i.test(column)) ||
columns.filter(column => /^[A-D]$/.test(column)).length >= 2;
});
});
let matchedPhase: string | undefined;
for (const start of anchors) {
if ((quotedProposal || contractCitation) && !currentContext(start)) continue;
if (sectionEvidence && (!sectionContext(tokens, start) || !activeSection(plain(tokens[start]!.raw)))) continue;
const anchor = tokens[start]!;
let end = start + 1;
while (end < tokens.length && !(tokens[end]!.type === 'heading' &&
(anchor.type !== 'heading' || (tokens[end] as any).depth <= anchor.depth))) end++;
// Keep a paragraph anchor's continuation (e.g. Header/Question) in
// the exact-field record, just as fields below a heading are retained.
const section = tokens.slice(anchor.type === 'paragraph' ? start : start + 1, end);
if (currentContext(start) && currentContext(tokens.indexOf(table))) {
// Reuse the owned ledger/source gates, but require one current row
// and comparison anchor before granting this exact-field path credit.
const currentRows = tokens.flatMap(token => {
if (token.type !== 'table' || !currentContext(tokens.indexOf(token))) return [];
const ids = token.header.flatMap((cell, i) => /^(?:ID|Decision)\b/i.test(plain(cell.text)) ? [i] : []);
return ids.length === 1 ? token.rows.map(row => plain(row[ids[0]!]!.text).split(/\s/, 1)[0]!.replace(/[.:]$/, '')) : [];
});
const ownedComparison = pendingRowContext(tokens, tokens.indexOf(table), read('id')) &&
sourceRecords.length <= 1 && sourceRecords.every(source => source === 'PLAN.md') &&
!hasForeignContractSource(cells[fields.evidence[0]!]!.text, sourcePlan) &&
currentRows.filter(value => value === id).length === 1 && recordAnchors.filter(currentContext).length === 1;
if (paragraphRecord(start) && (!pendingRowContext(tokens, tokens.indexOf(table), read('id')) ||
sourceRecords.some(source => source !== 'PLAN.md') ||
hasForeignContractSource(cells[fields.evidence[0]!]!.text, sourcePlan) ||
currentRows.filter(value => value === id).length !== 1 || recordAnchors.filter(currentContext).length !== 1)) continue;
if (ownedComparison && exactNativeFields(section)) matchedPhase = anchor.type === 'heading' ? plain(anchor.text) : plain(anchor.raw).split('\n')[0];
// Markdown permits an option paragraph followed by a facts list.
// Bind only the adjacent list to that option; never borrow a later
// option's facts, quoted/code content or another section's details.
const options = section.flatMap((token, index) => {
if (token.type === 'list') return token.items.map(item => proseOption(item.tokens));
if (token.type !== 'paragraph') return [];
let next = index + 1;
while (section[next]?.type === 'space') next++;
const details = section[next];
const facts = details?.type === 'list' && details.items.every(item => {
const first = item.tokens.find(part => part.type === 'text' || part.type === 'paragraph');
return first && /^(?:Effort|Risk|Pros|Cons|Reuse|Coverage)\s*:/i.test(plain(first.raw));
}) ? details.items.flatMap(item => item.tokens.filter(part => part.type === 'text' || part.type === 'paragraph')) : [];
return [proseOption([token, ...facts])];
}).filter(option => option !== null);
const baselineOption = (offered: string, saved: NonNullable<typeof options[number]>) => {
const addedAction = /(?:^|[.!?;]\s+|\n|[✅❌]\s*|\b(?:and|but|also|first|then|now|next|while)\s+)(?:please\s+)?(?:add(?:ing)?|remov(?:e|ing)|delet(?:e|ing)|cut(?:ting)?|drop(?:ping)?|replac(?:e|ing)|rewrit(?:e|ing)|chang(?:e|ing)|alter(?:ing)?|modif(?:y|ying)|enabl(?:e|ing)|disabl(?:e|ing)|implement(?:ing)?|install(?:ing)?|introduc(?:e|ing)|build(?:ing)?|writ(?:e|ing)|record(?:ing)?|captur(?:e|ing)|creat(?:e|ing)|switch(?:ing)?|migrat(?:e|ing)|externaliz(?:e|ing)|refactor(?:ing)?|expand(?:ing)?|reduc(?:e|ing)|deploy(?:ing)?|approv(?:e|ing)|run(?:ning)?)\b/i;
const unchanged = (action:string) => [saved.summary.slice(saved.label.length),q.options.find(o=>o.label===offered)?.description ?? ''].every(raw=>{
const text=optionText(raw), escaped=action.replace(/[.*+?^${}()|[\]\\]/g,'\\$&');
return current(text) && !addedAction.test(text) &&
!withdrawnOption.test(text) &&
!new RegExp(`\\b(?:not|never|no longer|doesn't|does not|will not)\\s+(?:(?:currently|now|actually)\\s+)?${escaped}(?:s|es)?\\b`,'i').test(text);
});
if (/\bvia\b/i.test(caption(offered)) !== /\bvia\b/i.test(caption(saved.label)) &&
/\bthrough\b/i.test(caption(offered)+' '+caption(saved.label)) && !unchanged(words(offered)[0] ?? '')) return false;
const baseline = savedBaseline(saved);
const offeredBaseline = savedBaseline({label:offered,bindingText:offered});
// Both captions explicitly retain this row's current baseline.
// A shortened action caption may omit its uniquely owned target;
// the current row must supply the whole native action and target,
// not another option, quoted history, a negation or a second match.
if (baseline && offeredBaseline && !baseline.generic && !offeredBaseline.generic) {
const short = baselineWords(baseline.caption), full = baselineWords(offeredBaseline.caption);
if (short.length && short.length < full.length && short.every((word,i)=>word===full[i])) {
const value=plain(cells[fields.current[0]!]!.text.replace(/"[^"\n]*"|“[^”\n]*”|(?<!\w)'[^'\n]*'|‘[^’\n]*’/g,''));
const valueWords=baselineWords(value);
const verb=(word:string)=>word===full[0] || word===full[0]+'s' || (full[0]!.endsWith('s') && word===full[0]+'es');
const hits=valueWords.flatMap((word,i)=>verb(word) && full.slice(1).every((next,j)=>valueWords[i+j+1]===next)?[i]:[]);
// A one-word action caption can omit an explicit destination
// and numeric outcome. Both must occur together in Current;
// the same-letter grid must name that action and retain the
// exact outcome. This is not unordered word-overlap matching.
const result = /^([a-z][a-z0-9_]*)\s+([0-9]+)$/i.exec(offeredBaseline.caption.split(',').at(-1)!.trim());
const operands = valueWords.flatMap((_, i) => full.slice(1).every((word, j) => valueWords[i+j] === word) ? [i] : []);
const explicitOutcome = short.length === 1 && /^(?:to|from|through|via)$/.test(full[1] ?? '') &&
selector(offered) === selector(saved.label) && result && operands.length === 1 &&
[saved.summary.slice(saved.label.length), q.options.find(o => o.label === offered)?.description ?? ''].every(raw =>
[...optionText(raw).matchAll(new RegExp(`\\b${result[1]}\\s+([0-9]+)\\b`, 'gi'))]
.every(claim => claim[1] === result[2])) && section.some(token => {
if (token.type !== 'table' || !currentContext(tokens.indexOf(token))) return false;
const headers = token.header.map(cell => plain(cell.text)), ids = headers.map(selector);
const own = ids.indexOf(selector(offered)), currentColumn = headers.findIndex(h => /^Current$/i.test(h));
const commitment = headers.findIndex(h => /^Commitment$/i.test(h));
if (headers.length !== q.options.length + 3 || own < 0 || currentColumn < 0 || commitment < 0 ||
!headers.some(h => /^Source(?:\b|\/)/i.test(h)) ||
!q.options.every(o => ids.filter(id => id === selector(o.label)).length === 1) ||
!sameBaseline(retainedCaption(headers[own]!), baseline.caption)) return false;
const rows = token.rows.filter(row => plain(row[commitment]!.text).split(/\s/, 1)[0]!.toLowerCase() === result[1]!.toLowerCase());
return rows.length === 1 && plain(rows[0]![currentColumn]!.text) === result[2] && plain(rows[0]![own]!.text) === result[2];
});
return namedSource && currentContext(tokens.indexOf(table)) && current(value) &&
!/\b(?:not|never|no longer|[a-z]+n['’]t|will|would|could|should|may|might|previously|formerly|historical|hypothetical|if|unless|withdrawn|retracted|superseded|(?:other|another|foreign) (?:plan|project))\b/i.test(value) &&
(!selector(offered) || selector(offered)===selector(saved.label)) &&
(hits.length===1 || explicitOutcome) && unchanged(full[0]!);
}
}
if (!baseline || (!baseline.generic && !/^(?:keep|retain|preserve)\b/i.test(baselineCaption(offered))))
return sameOption(offered, saved.label, selector(offered) ? saved.summary : saved.bindingText);
const offeredId = selector(offered), savedId = selector(saved.label);
if (offeredId && offeredId !== savedId) return false;
const retained = retainedCaption(offered);
if (!baseline.generic) {
if (!sameBaseline(retained, baseline.caption)) return false;
// The concrete caption itself identifies the unchanged plan
// alternative in this source-bound row's complete comparison.
return baselineWords(baseline.caption).length >= 2;
}
if (!offeredId || offeredId !== savedId || (baseline.caption && !sameBaseline(retained, baseline.caption))) return false;
const alternatives = [...read('proposed').matchAll(/(?:^|\s)([A-D])[).:]\s+(.+?)(?=\s[A-D][).:]\s|$)/g)];
const own = alternatives.filter(match => match[1] === savedId);
if (own.length !== 1 || !sameBaseline(retained, own[0]![2]!)) return false;
// A generic caption is resolved by the same-letter Proposed
// alternative AND its unchanged Current column in the owned grid.
// The complete prose option still owns effort/risk/pros/cons.
return section.some(token => {
if (token.type !== 'table' || !currentContext(tokens.indexOf(token))) return false;
const headers = token.header.map(cell => plain(cell.text));
const identity = (header: string) => /^[A-D]$/.test(header) ? header : selector(header);
const ids = headers.map(identity), baselineIndex = ids.indexOf(savedId);
const currentIndex = headers.findIndex(header => /^Current$/i.test(header));
const contractIndex = headers.findIndex(header => /^Commitment$/i.test(header));
const sourceIndex = headers.findIndex(header => /^Source(?:\b|\/)/i.test(header));
const optionIndices = ids.flatMap((id, i) => id ? [i] : []);
if (headers.length !== q.options.length + 3 || optionIndices.length !== q.options.length ||
new Set(optionIndices.map(i => ids[i])).size !== q.options.length || baselineIndex < 0 ||
currentIndex < 0 || contractIndex < 0 || sourceIndex < 0) return false;
const rawRows = token.raw.trimEnd().split('\n').slice(2);
const rows = token.rows.map(row => row.map(cell => plain(cell.text)));
return rows.length > 0 && rows.every((row, index) => /(^|[^\\])\|/.test(rawRows[index] ?? '') &&
row.length === headers.length && row.every(cell => cell && current(cell)) &&
row[baselineIndex]!.toLowerCase() === row[currentIndex]!.toLowerCase()) &&
rows.some(row => optionIndices.some(i => row[i]!.toLowerCase() !== row[currentIndex]!.toLowerCase()));
});
};
const matched = q.options.map(offered => options.flatMap((saved, index) =>
saved!.complete && baselineOption(offered.label, saved!) ? [index] : []));
if (options.length === q.options.length && matched.every(found => found.length === 1) &&
new Set(matched.flat()).size === q.options.length) {
matchedPhase = anchor.type === 'heading' ? plain(anchor.text) : plain(anchor.raw).split('\n')[0];
}
}
for (const comparison of tokens.slice(start + 1, end)) {
if (comparison.type !== 'table') continue;
if ((quotedProposal || contractCitation) && !currentContext(tokens.indexOf(comparison))) continue;
if (sectionEvidence && !sectionContext(tokens, tokens.indexOf(comparison))) continue;
const headers = comparison.header.map(c => plain(c.text));
// The declared commitment grid transposes the option table: each
// complete alternative is a column. Its saved effort/risk row and
// behavioral cells bind the owned native pros/cons for that option.
const commitment = headers.findIndex(h => /^Commitment$/i.test(h));
const source = headers.findIndex(h => /^Source(?:\b|\/)/i.test(h));
const baseline = headers.findIndex(h => /^Current$/i.test(h));
const gridSelector = (header: string) => selector(header) ?? /^([A-D])$/i.exec(header)?.[1]?.toUpperCase();
const optionColumns = headers.flatMap((header, index) => gridSelector(header) ? [index] : []);
if (commitment >= 0 && source >= 0 && baseline >= 0 &&
currentContext(start) && currentContext(tokens.indexOf(table)) && currentContext(tokens.indexOf(comparison)) &&
headers.length === q.options.length + 3 &&
optionColumns.length === q.options.length &&
new Set(optionColumns.map(i => gridSelector(headers[i]!))).size === q.options.length) {
// GFM permits a following un-delimited paragraph as a padded row.
// Only explicit grid rows supply cells; a current prose footer
// remains context and cannot fill a missing value in a real row.
const rawRows = comparison.raw.trimEnd().split('\n').slice(2);
const gridRow = (index: number) => /(^|[^\\])\|/.test(rawRows[index] ?? '');
const footerCurrent = rawRows.filter((_, index) => !gridRow(index)).every(line => current(plain(line)));
const rows = comparison.rows.filter((_, index) => gridRow(index)).map(row => row.map(cell => plain(cell.text)));
const effortRisk = rows.filter(row => /^Effort\s*\/\s*risk$/i.test(row[commitment] ?? ''));
const effort = rows.filter(row => /^Effort$/i.test(row[commitment] ?? ''));
const risk = rows.filter(row => /^Risk$/i.test(row[commitment] ?? ''));
const behavior = rows.filter(row => !/^(?:Effort(?:\s*\/\s*risk)?|Risk)$/i.test(row[commitment] ?? ''));
const scalar = (value: string, kind: 'effort' | 'risk') => {
const match = /^(S|M|L|XL|low|medium|high)(?:\s*\(([^()]*)\))?$/i.exec(value);
return Boolean(match && (kind === 'effort' ? /^(?:S|M|L|XL)$/i : /^(?:low|medium|high)$/i).test(match[1]!) &&
(kind !== 'effort' || !(match[2]?.match(/\b(?:S|M|L|XL)\b/gi) ?? []).some(size => size.toUpperCase() !== match[1]!.toUpperCase())) &&
current(match[2] ?? '') && !/\b(?:not|never|no longer|withdrawn|retracted|superseded|historical|previously|formerly|low|medium|high|risk|effort)\b/i.test(match[2] ?? ''));
};
const separate = effortRisk.length === 0 && effort.length === 1 && risk.length === 1;
const metadata = separate ? optionColumns.every(i => scalar(effort[0]![i] ?? '', 'effort') && scalar(risk[0]![i] ?? '', 'risk'))
: effortRisk.length === 1 && effort.length === 0 && risk.length === 0 && optionColumns.every(i => /^(?:S|M|L|XL)\s*\/\s*(?:low|medium|high)$/i.test(effortRisk[0]![i] ?? ''));
const bare = optionColumns.some(i => /^[A-D]$/i.test(headers[i]!));
const declarations = [...read('proposed').matchAll(/(?:^|[.;]\s+)([A-D])[).:]\s+(.+?)(?=[.;]\s+[A-D][).:]\s+|$)/g)]
.map(match => `${match[1]}) ${match[2]}`);
const uniqueGrid = !(bare || separate) || (sourceRecords.length <= 1 && sourceRecords.every(source => source === 'PLAN.md') &&
anchors.length === 1 && section.filter(token => token.type === 'table' &&
token.header.some(cell => /^Commitment$/i.test(plain(cell.text)))).length === 1);
const complete = footerCurrent && metadata && uniqueGrid &&
behavior.length > 0 && behavior.every(row => row.length === headers.length && row[commitment] && row[source] && row[baseline] &&
current(row[commitment]!) && (!(bare || separate) || (current(row[source]!) && !hasForeignContractSource(row[source]!, sourcePlan))) &&
optionColumns.every(i => row[i] && current(row[i]!))) &&
behavior.some(row => new Set(optionColumns.map(i => row[i]!.toLowerCase())).size > 1) &&
q.options.every(o => { const facts = prose(o.description ?? ''); return /✅/.test(facts) && /❌/.test(facts) && current(facts) &&
(!(bare || separate) || !withdrawnOption.test(optionText(facts))); });
const matched = q.options.map(offered => optionColumns.filter(i => /^[A-D]$/i.test(headers[i]!)
? selector(offered.label) === gridSelector(headers[i]!) && declarations.length === q.options.length &&
declarations.filter(saved => selector(saved) === gridSelector(headers[i]!) && declaredOption(offered, saved)).length === 1
: completeCaption(offered.label, headers[i]!, '')));
if (complete && matched.every(found => found.length === 1) && new Set(matched.flat()).size === q.options.length)
matchedPhase = anchor.type === 'heading' ? plain(anchor.text) : plain(anchor.raw).split('\n')[0];
}
const optionColumn = headers.findIndex(h => /^(?:Option|Approach)\b/i.test(h));
if (optionColumn < 0 || !['effort', 'risk', 'pros', 'cons'].every(h => headers.some(v => v.toLowerCase() === h)) ||
comparison.rows.length !== q.options.length || comparison.rows.some(row => row.some(cell => !plain(cell.text)))) continue;
const saved = comparison.rows.map(row => plain(row[optionColumn]!.text));
const summaryColumn = headers.findIndex(h => /^(?:Summary|Description|Approach)$/i.test(h));
const matched = q.options.map(offered => saved.flatMap((label, i) => sameOption(offered.label, label,
summaryColumn < 0 ? '' : plain(comparison.rows[i]![summaryColumn]!.text)) ? [i] : []));
if (matched.every(found => found.length === 1) && new Set(matched.flat()).size === q.options.length) {
matchedPhase = anchor.type === 'heading' ? plain(anchor.text) : plain(anchor.raw).split('\n')[0];
}
}
}
if (matchedPhase) matches.push({ ledgerId: id, phase: matchedPhase });
}
}
return matches.length === 1 ? matches[0]! : null;
}
/** Fixture-local metric adapter. It never advances the shared phase boundary.
* Every real current question, including repeated remedies, still counts
* toward the original 4–7 band. Unknown decisions fail closed. */
export function createCeoPaymentFindingCounter(seedPlan: string, readPlan: () => string,
existingFinding: (fp: AskUserQuestionFingerprint) => boolean) {
const trace: Array<Finding | { signature: string; kind: 'setup' | 'existing-finding' | 'additional-current-decision' } |
{ signature: string; kind: 'recorded-decision'; ledgerId: string; phase: string }> = [];
return {
trace,
isReviewAUQ(fp: AskUserQuestionFingerprint, priorCalls: readonly NativePlanQuestionCall[] = []): boolean {
const setupPacket = ownedSetupPacket(fp);
if ((!ownedAnswer(fp) && !setupPacket) || priorCalls.some(call => `${call.sessionId}:${call.toolUseId}` === fp.signature))
throw new Error(`Invalid or duplicated completed native decision: ${fp.signature}`);
if (setupPacket || setupQuestion(fp)) { trace.push({ signature: fp.signature, kind: 'setup' }); return false; }
const plan = readPlan();
const finding = ceoPaymentFinding(fp, seedPlan, plan);
if (finding) { trace.push(finding); return true; }
const decision = recordedDecision(fp, plan, seedPlan);
if (decision) { trace.push({ signature: fp.signature, kind: 'recorded-decision', ...decision }); return true; }
if (todoDecision(fp)) { trace.push({ signature: fp.signature, kind: 'additional-current-decision' }); return true; }
if (existingFinding(fp)) { trace.push({ signature: fp.signature, kind: 'existing-finding' }); return true; }
const q = fp.nativeCall!.questions[0]!, predicates: string[] = [];
ceoPaymentFinding(fp, seedPlan, plan, predicates);
throw new Error(`Unsupported current CEO decision; cannot exclude it from the 4–7 count: ${fp.signature}\n` +
`header: ${q.header}\nquestion: ${q.question.slice(0, 200)}\n` +
`obligation predicates: ${predicates.length ? predicates.join('; ') : 'no active PLAN.md-sourced ledger row named by the question shares an obligation subject'}`);
},
};
}
File diff suppressed because it is too large. Load diff
+6 -175
View File
@@ -40,8 +40,6 @@ import {
stripAnsi,
auqFingerprint,
COMPLETION_SUMMARY_RE,
MODE_RE,
findModeOption,
classifyPlanCountFrame,
capturePlanCountQuestion,
matchesNativePlanQuestion,
@@ -54,7 +52,6 @@ import {
engSetupAUQ,
engFirstReviewAUQ,
designStep0Boundary,
designFirstReviewAUQ,
planCountQuestionPhase,
nativePlanCallFingerprint,
devexStep0Boundary,
@@ -88,74 +85,6 @@ describe('saved preference annotation', () => {
});
describe('mode option rendering', () => {
test('letter-prefixed native mode labels retain their actual target indices', () => {
const options = ['C — HOLD SCOPE (Recommended)', 'B — SELECTIVE EXPANSION', 'A — SCOPE EXPANSION', 'D — SCOPE REDUCTION']
.map((label, i) => ({ index: i + 1, label }));
expect(options.every(option => MODE_RE.test(option.label))).toBe(true);
for (const [mode, index] of [['HOLD SCOPE', 1], ['SELECTIVE EXPANSION', 2], ['SCOPE EXPANSION', 3], ['SCOPE REDUCTION', 4]] as const) {
expect(findModeOption(options, mode)?.index).toBe(index);
}
expect(findModeOption(options.filter(option => option.index !== 3), 'SCOPE EXPANSION')).toBeUndefined();
});
test('letter-prefixed matching excludes prose, unrelated choices and unsupported framing', () => {
for (const label of ['Choose C — HOLD SCOPE', 'Approach C — HOLD SCOPE', 'C — Keep this approach\nHOLD SCOPE',
'B — Ideal Architecture (Recommended)', 'A — Fix-Only (Minimal Viable)', 'CC — HOLD SCOPE', 'E — HOLD SCOPE',
'1 — HOLD SCOPE', 'C: HOLD SCOPE', 'C - HOLD SCOPE', 'C — SCOPE EXPANSIONIST']) {
expect(MODE_RE.test(label), label).toBe(false);
expect(findModeOption([{ index: 1, label }], 'HOLD SCOPE'), label).toBeUndefined();
}
expect(findModeOption([{ index: 1, label: 'C — HOLD SCOPE\nPrefer this over A — SCOPE EXPANSION.' }], 'SCOPE EXPANSION')).toBeUndefined();
});
test('parenthesized native modes preserve capture group and actual target indices', () => {
const options = ['A) SCOPE EXPANSION', 'B) SELECTIVE EXPANSION (recommended)', 'C) HOLD SCOPE', 'D) SCOPE REDUCTION']
.map((label, i) => ({ index: i + 1, label }));
for (const [mode, index] of [['SCOPE EXPANSION', 1], ['SELECTIVE EXPANSION', 2], ['HOLD SCOPE', 3], ['SCOPE REDUCTION', 4]] as const) {
expect(MODE_RE.exec(options[index - 1]!.label)?.[1]).toBe(mode);
expect(findModeOption(options, mode)?.index).toBe(index);
}
expect(findModeOption(options.filter(option => option.index !== 1), 'SCOPE EXPANSION')).toBeUndefined();
expect(findModeOption([{ index: 3, label: '**C) HOLD SCOPE**' }], 'HOLD SCOPE')?.index).toBe(3);
});
test('parenthesized mode recognition rejects prose, descriptions and unrelated framing', () => {
for (const label of ['Discuss C) HOLD SCOPE', 'Approach C) HOLD SCOPE', 'C) Keep this approach\nHOLD SCOPE',
'B) Ideal Architecture (Recommended)', 'A) Fix-Only (Minimal Viable)', 'CC) HOLD SCOPE', 'E) HOLD SCOPE',
'1) HOLD SCOPE', '(C) HOLD SCOPE', 'C: HOLD SCOPE', 'C - HOLD SCOPE', 'C) SCOPE EXPANSIONIST']) {
expect(MODE_RE.test(label), label).toBe(false);
expect(findModeOption([{ index: 1, label }], 'HOLD SCOPE'), label).toBeUndefined();
}
expect(findModeOption([{ index: 1, label: 'C) HOLD SCOPE\nPrefer this over A) SCOPE EXPANSION.' }], 'SCOPE EXPANSION')).toBeUndefined();
});
test('selects the actual collapsed-space mode from the failed periodic menu', () => {
const options = [
{ index: 1, label: 'HOLDSCOPE(recommended)\rMake the reliability wave bulletproof.' },
{ index: 2, label: 'SELECTIVEEXPANSION\rKeep the current scope as the baseline.' },
{ index: 3, label: 'SCOPEREDUCTION\rFind the minimum subset.' },
{ index: 4, label: 'SCOPEEXPANSION\rDream up adjacent reliability improvements.' },
{ index: 5, label: 'Type something.' },
{ index: 6, label: 'Chat about this\rUser answered → HOLD SCOPE (recommended)' },
];
expect(options.slice(0, 4).every(option => MODE_RE.test(option.label))).toBe(true);
expect(findModeOption(options, 'SCOPE EXPANSION')?.index).toBe(4);
expect(findModeOption(options, 'HOLD SCOPE')?.index).toBe(1);
});
test('retains ordinary, wrapped, and emphasized mode labels', () => {
for (const label of ['SCOPE EXPANSION (Recommended)', 'scope\t expansion', 'SCOPE\r\nEXPANSION', '**SCOPE EXPANSION**']) {
expect(findModeOption([{ index: 2, label }], 'SCOPE EXPANSION')?.index).toBe(2);
}
});
test('an omitted target remains missing, including when another mode mentions it', () => {
const options = [
{ index: 1, label: 'HOLD SCOPE\rPrefer this over SCOPE EXPANSION.' },
{ index: 2, label: 'SELECTIVE EXPANSION' },
{ index: 3, label: 'SCOPE REDUCTION' },
{ index: 4, label: 'Chat about this\rSCOPEEXPANSION (old screen)' },
];
expect(findModeOption(options, 'SCOPE EXPANSION')).toBeUndefined();
expect(MODE_RE.test(options[3]!.label)).toBe(false);
expect(MODE_RE.test('Scope expansionist')).toBe(false);
});
});
describe('isPermissionDialogVisible', () => {
@@ -297,6 +226,12 @@ describe('isPermissionDialogVisible', () => {
// post-merge follow-up. Flip this assertion once the regex tightens.
expect(isPermissionDialogVisible(sample)).toBe(true);
});
test('matches the captured Autoplan settings-overwrite card as a numbered permission dialog', () => {
const captured = JSON.parse(readFileSync(new URL('../fixtures/autoplan-settings-overwrite.json', import.meta.url), 'utf8'));
expect(isNumberedOptionListVisible(captured.frame.text)).toBe(true);
expect(isPermissionDialogVisible(captured.frame.text)).toBe(true);
});
});
describe('isNumberedOptionListVisible', () => {
@@ -1207,10 +1142,6 @@ describe('parseQuestionPrompt', () => {
expect(prompt).toStartWith('Reviewfocus');
expect(prompt).toContain('design completeness');
expect(prompt).not.toContain('Planning:');
expect(designStep0Boundary({
signature: 'captured-design-scope', promptSnippet: prompt,
options: parseNumberedOptions(visible), observedAtMs: 0, preReview: true,
})).toBe(true);
});
test('keeps the captured devex persona header when cursor spacing collapses', () => {
@@ -1222,10 +1153,6 @@ describe('parseQuestionPrompt', () => {
].join('\n');
const prompt = parseQuestionPrompt(visible);
expect(prompt).toStartWith('Targetpersona');
expect(devexStep0Boundary({
signature: 'captured-devex-persona', promptSnippet: prompt,
options: parseNumberedOptions(visible), observedAtMs: 0, preReview: true,
})).toBe(true);
});
test('retains a multiline question while excluding the preceding CLI divider', () => {
@@ -2106,54 +2033,6 @@ describe('Step0BoundaryPredicate per-skill', () => {
fingerprint.promptSnippet = question.slice(0, 240);
return fingerprint;
};
test('FIRES on the current template Step 0D focus-area question', () => {
expect(focusTemplate).toContain('Want me to focus on specific areas instead of all 7?');
const question = focusQuestion('hierarchy, spacing, contrast');
expect(question.length).toBeLessThanOrEqual(240);
expect(designStep0Boundary(fp(question, focusOptions))).toBe(true);
});
test('reads the owned full focus question when its gap list exceeds the diagnostic snippet', () => {
const question = focusQuestion('primary-action hierarchy, inconsistent vertical rhythm, inaccessible error contrast, label-size drift, absent loading feedback, and missing recovery states');
const fingerprint = nativeFocus(question);
expect(fingerprint.promptSnippet).not.toContain('Want me to focus');
expect(question.length).toBeGreaterThan(240);
expect(designStep0Boundary(fingerprint)).toBe(true);
});
test.each([
"I've rated this plan 4/10 on design completeness. Should we add a loading state?",
'Want me to focus on specific areas instead of all 7?',
"I've rated the error message 4/10 on design completeness. Want me to focus on specific areas instead of all 7?",
"I've rated this plan 4/10 on design completeness. Should we focus on correcting error contrast?",
])('does NOT turn a later finding into setup from a partial focus match: %s', question => {
expect(designStep0Boundary(nativeFocus(question))).toBe(false);
});
test('does NOT combine partial focus matches across separate native question tabs', () => {
const fingerprint = nativeFocus("I've rated this plan 4/10 on design completeness. Should we add a loading state?");
const call = fingerprint.nativeCall!;
const secondQuestion = 'Want me to focus on specific areas instead of all 7?';
call.questions.push({ ...call.questions[0]!, question: secondQuestion });
call.answers![secondQuestion] = focusOptions[0]!;
expect(designStep0Boundary(fingerprint)).toBe(false);
});
test('FIRES on design system / posture mention', () => {
const f = fp('Pick a design posture for this review', ['Polish', 'Triage', 'Expansion']);
expect(designStep0Boundary(f)).toBe(true);
});
test('FIRES on first-dimension prompt', () => {
const f = fp('First dimension: visual hierarchy. Score?', ['7', '8', '9']);
expect(designStep0Boundary(f)).toBe(true);
});
test('does NOT fire on later dimension AUQs', () => {
const f = fp('Spacing dimension score?', ['7', '8', '9']);
expect(designStep0Boundary(f)).toBe(false);
});
});
describe('design review begins without an optional focus question', () => {
@@ -2168,57 +2047,9 @@ describe('Step0BoundaryPredicate per-skill', () => {
'☐DEIGN.md TODO │D6 — TODO: Create a DESIGN.md file codifying the5 decisions mdein this revew <gstack-qid:plan-design-review-todo-designmd>',
'☐PartialfailTODO │D7—TODO:Specifythepartial-failurestate—whatdoestheuserseeifSavesucceedsforsomefieldsbutfailsfor others? <gstack-qid:plan-design-review-todo-partialfail>',
];
test('counts the first captured finding and every subsequent finding', () => {
let reviewStarted = false;
const phases = questions.map(question => {
const phase = planCountQuestionPhase(fp(question, ['Apply', 'Defer']), reviewStarted,
designStep0Boundary, designFirstReviewAUQ);
reviewStarted = phase.reviewStarted;
return phase.preReview;
});
expect(phases).toEqual([false, false, false, false, false, false, false]);
});
test('keeps the observed focus gate separate when it is emitted', () => {
const focus = fp("☐ Focus areas │ I've rated this plan2/10 on design completeness. Review all7 dimensions?", ['All7dimensions', 'Priority gaps']);
const setup = planCountQuestionPhase(focus, false, designStep0Boundary, designFirstReviewAUQ);
expect(setup).toEqual({ preReview: true, reviewStarted: true });
expect(planCountQuestionPhase(fp(questions[0], ['Apply', 'Defer']), setup.reviewStarted,
designStep0Boundary, designFirstReviewAUQ)).toEqual({ preReview: false, reviewStarted: true });
});
test('requires review identity, not just a D1 label or setup question ID', () => {
for (const question of [
'☐ Setup │D1—Enable cross-project learnings?',
'☐ Review target │D1—Which plan should I review?<gstack-qid:plan-design-review-scope>',
'☐ Focus │D1—What should this design review focus on?<gstack-qid:plan-design-review-focus-areas>',
'☐ Scope │I will review Pass1 through Pass7 after setup. Proceed?',
]) expect(designFirstReviewAUQ(fp(question, ['Yes', 'No']))).toBe(false);
expect(designFirstReviewAUQ(fp('☐ Page structure │ Pass1 — Information Architecture: what page structure should this use?', ['Standard', 'Sidebar']))).toBe(true);
});
test('leaves callers without a first-review predicate unchanged', () => {
expect(planCountQuestionPhase(fp(questions[0], ['Apply', 'Defer']), false, designStep0Boundary))
.toEqual({ preReview: true, reviewStarted: false });
});
});
describe('devexStep0Boundary', () => {
test('FIRES on developer persona selection', () => {
const f = fp('Pick the target persona for this review', ['Senior backend', 'Junior frontend', 'Other']);
expect(devexStep0Boundary(f)).toBe(true);
});
test('FIRES on TTHW target prompt', () => {
const f = fp('What is the TTHW target for first run?', ['<5 min', '<15 min', '<30 min']);
expect(devexStep0Boundary(f)).toBe(true);
});
test('does NOT fire on review-section AUQs', () => {
const f = fp('Friction point: 5-min CI wait. Address?', ['Now', 'Defer', 'Skip']);
expect(devexStep0Boundary(f)).toBe(false);
});
});
});
+1 -7
View File
@@ -1,6 +1,6 @@
/** Private evaluator inputs. Never copy this module or its JSON output into producer snapshots. */
import { createHash } from 'node:crypto';
import { CORPUS_VERSION, type EvalFamily, type EvalStack } from '../fixtures/cso-eval/materialize';
import { CORPUS_VERSION, type EvalFamily } from '../fixtures/cso-eval/materialize';
export interface OracleRequest { method: 'GET' | 'POST'; path: string; body?: string; headers?: Record<string, string> }
export interface OracleResponse { status: number; body: string; headers?: Record<string, string> }
@@ -100,9 +100,3 @@ export function judgeRepair(family: EvalFamily, evidence: PrivateEvidence): { re
&& patched.existingTestsPassed && evidence.immutableVerifier && evidence.independentRootCauseReview && evidence.featurePreserved && !evidence.boundaryMocks;
return { reproduced, correctRepair, evidenceHash: createHash('sha256').update(JSON.stringify({ version: oracle.version, family, evidence })).digest('hex') };
}
export function runtimeStart(stack: EvalStack): { executable: string; args: string[]; port: 8000; environment: Record<string, string> } {
return stack === 'node' ? { executable: '/usr/local/bin/node', args: ['app.mjs'], port: 8000, environment: {} }
: stack === 'bun' ? { executable: '/usr/local/bin/bun', args: ['--no-install', 'app.ts'], port: 8000, environment: {} }
: stack === 'python' ? { executable: '/usr/local/bin/python', args: ['-I', 'app.py'], port: 8000, environment: {} }
: { executable: '/usr/local/bin/ruby', args: ['bin/rails', 'server', '-e', 'test', '-b', '127.0.0.1', '-p', '8000'], port: 8000, environment: { RAILS_ENV: 'test', RACK_ENV: 'test' } };
}
-91
View File
@@ -1,91 +0,0 @@
import type { AskUserQuestionFingerprint } from './claude-pty-runner';
/** Recording or skipping an explicitly deferred typography TODO preserves the
* current design. Selecting its build-now alternative remains a review choice.
*/
function deferredTypographyTodo(q: NonNullable<AskUserQuestionFingerprint['nativeCall']>['questions'][number], selected: string): boolean {
const clean = (text: string) => text.trim().replace(/\s+/g, ' ');
const lines = q.question.trim().split('\n').map(clean);
const title = /^D[1-9]\d* [—–-] TODO proposal: (?:record|add) a deferred TODOS\.md (?:item|note) to (?:evaluate|explore|consider) [^?\n]+ \(replacing ([A-Za-z][A-Za-z0-9-]{0,39})\) (?:in a later|during a future) design pass\?$/.exec(lines[0] ?? '');
if (!title || !/^TODO [1-9]\d*$/.test(q.header) || lines.length !== 7 ||
!/^Project\/branch\/task: [A-Za-z0-9_./-]+, \/plan-design-review of PLAN\.md, post-pass TODOS\.md updates\.$/.test(lines[1]!)) return false;
const font = title[1]!;
const assessment = lines[2]!;
// Bind the current scope before reading the optional explanation of future
// value. Font examples, effort estimates and brand prose are not evidence.
const scope = new RegExp(`^ELI10: DESIGN\\.md and (?:this|the current) plan (?:keep|retain|preserve) ${font} as the app font(?:, and you excluded visual exploration from this update|\\. Visual exploration (?:remains|is) out of scope for this update), so (?:nothing changes now|the current design remains unchanged)\\.`);
const recordOnly = /\bThis question is only about whether to (?:write|record) [^.]+ in TODOS\.md [^.]*\bfuture \/design-consultation\b[^.]*, not about changing anything here\./.test(assessment)
|| /\bThis question only (?:records|skips) a deferred TODOS\.md (?:item|note) for a future \/design-consultation; it does not change the current design\./.test(assessment);
if (!scope.test(assessment) || !recordOnly ||
!/^Stakes if we pick wrong: .+\.$/.test(lines[3]!) ||
!/^Recommendation: A\b/.test(lines[4]!) || !/\bout[- ]of[- ]scope\b/.test(lines[4]!) ||
!/^Note: options differ in kind, not coverage\b/.test(lines[5]!) ||
!/^Net: .+\.$/.test(lines[6]!)) return false;
const choices = ['A Add to TODOS.md', 'B Skip, not valuable enough', 'C Build it now in this PR'];
const labels = q.options.map(o => clean(o.label).replace(/ \(recommended\)$/i, ''));
if (labels.length !== 3 || choices.some(label => labels.filter(l => l === label).length !== 1)) return false;
const description = (label: string) => clean(q.options[labels.indexOf(label)]!.description!);
const [add, skip, build] = choices.map(description) as [string, string, string];
const selectedLabel = labels[q.options.findIndex(o => o.label === selected)];
if (selectedLabel !== choices[0] && selectedLabel !== choices[1]) return false;
// A and B are record/skip-only choices; C is explicitly an implementation
// alternative. Additional present-work instructions invalidate the boundary.
const all = [...lines, add, skip, build].join('\n');
if (/\b(?:Correction|Hypothetical|Source only|Example only)\s*:|\b(?:this (?:scope|proposal|deferment)|the (?:scope|proposal|deferment)) (?:is|was) (?:withdrawn|cancelled|rejected)\b/i.test(all) ||
/\b(?:Also|Additionally|Instead|Now|Then)\s+(?:we\s+)?(?:must\s+|will\s+)?(?:fix|replace|change|implement|build|add|load|remove)\b/i.test(all) ||
/\b(?:font replacement|visual exploration|typography work)\s+(?:is|becomes|remains)\s+(?:now\s+)?in scope\b/i.test(all)) return false;
const unquoted = (text: string) => text.replace(/"[^"\n]*"|“[^”\n]*”/g, '[quoted]');
const retained = unquoted([...lines, add, skip].join('\n'));
if (/\bVisual exploration (?:is|remains) (?:no longer|not) out of scope\b/i.test(retained) ||
new RegExp(`\\b(?:This|The current) plan (?:no longer|does not) (?:keeps?|retains?|preserves?) ${font}\\b`, 'i').test(retained)) return false;
const recording = unquoted(add + '\n' + skip).split(/[.!?]\s+|[✅❌]|\n/).map(clean);
if (recording.some(sentence => /^(?:Add|Create|Fix|Replace|Change|Implement|Build|Load|Remove|Set|Make)\b/i.test(sentence) &&
(!/^(?:Add|Create) (?:a |the |one )?TODOS\.md (?:file|item|note)(?: for (?:the )?(?:future|deferred|later) [A-Za-z0-9 /_-]+)?\.?$/i.test(sentence) || /\b(?:and|then|plus)\s+(?:add|create|fix|replace|change|implement|build|load|remove|set|make)\b/i.test(sentence)))) return false;
if (/\b(?:it is false that|not true that|if approved|no longer preserves?)\b/i.test(retained) ||
/\b(?:replace|change|implement|build|load|remove)\b[^.!?\n]*\b(?:now|in this PR|in this update)\b/i.test(retained.replace(/not about changing anything here\./g, '').replace(/do not add it now/g, 'deferred'))) return false;
const preservesFont = new RegExp(`(?:Nothing changes|No design changes) in this update; DESIGN\\.md and ${font} (?:stay as approved|remain unchanged)\\.`);
return /\b(?:next|future) \/design-consultation\b/.test(add) && preservesFont.test(add) &&
/\b(?:Adds a TODOS\.md file|Records only a TODOS\.md note)\b/.test(add) &&
/\b(?:No TODOS\.md noise|No TODO is recorded)\b/.test(skip) && /\b(?:Zero follow-up work|No follow-up work)\./.test(skip) &&
/\b(?:immediately|now|this PR)\b/i.test(build) && /\b(?:Fonts? load|Replace the font|Change the font)\b/i.test(build);
}
/** Accepted rendering of existing decisions adds an artifact, not a finding.
* It still changes the deliverable and therefore remains a freshness boundary.
*/
export function isDesignArtifactGeneration(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.sessionId || !call.toolUseId || call.answered !== true || call.failed !== false ||
call.questions.length !== 1 || !Array.isArray(call.unansweredQuestionIndices) ||
call.unansweredQuestionIndices.length || !Number.isFinite(Date.parse(call.answeredAt ?? '')) ||
fp.signature !== `${call.sessionId}:${call.toolUseId}`) return false;
const q = call.questions[0]!;
if (q.multiSelect || q.options.length < 2 || q.options.length > 3 || Object.keys(call.answers ?? {}).length !== 1 ||
q.options.some(o => typeof o.description !== 'string' || ('preview' in o && Boolean(o.preview))) ||
fp.options.length !== q.options.length || fp.options.some((o, i) => o.index !== i + 1 || o.label !== q.options[i]!.label)) return false;
const clean = (text: string) => text.trim().replace(/\s+/g, ' ');
const label = (text: string) => clean(text).replace(/^[AB]\) /, '').replace(/ \(Recommended\)$/, '');
const positive = q.options.find(o => call.answers?.[q.question] === o.label);
if (!positive) return false;
if (q.options.length === 3) return (fp.nativeQuestionIndex === undefined || fp.nativeQuestionIndex === 0) &&
deferredTypographyTodo(q, positive.label);
const other = q.options.find(o => o !== positive)!;
const question = clean(q.question);
const description = clean(positive.description!);
const alternative = clean(other.description!);
// Consume every sentence. A heading or "no new decisions" claim alone
// cannot conceal an added requirement, omitted state, or actual design choice.
if (/^D\d+ StateTable$/.test(q.header) &&
/^D\d+ — Add a state coverage table to the plan body for implementer reference\? <gstack-qid:plan-design-review-states-\d+>$/.test(question)) {
return label(positive.label) === 'Add state table' && label(other.label) === 'Leave states in prose only' &&
/^Insert a feature × state table \(Form load \/ Save \/ Export \/ Dirty state × Loading \/ Empty \/ Error \/ Success \/ Pending\)\. No new design decisions — all cells derive from existing specs\. Completeness: \d+\/10 — implementers can verify each state against a single reference\.$/.test(description) &&
/^Keep the existing prose descriptions without a structured table\. Completeness: \d+\/10 — specs are all there but scattered across paragraphs; edge cases like Export error during dirty-edit are harder to spot\.$/.test(alternative);
}
if (/^D\d+ Storyboard$/.test(q.header) &&
/^D\d+ — Add a user journey storyboard to the plan\? <gstack-qid:plan-design-review-journey-\d+>$/.test(question)) {
return label(positive.label) === 'Add storyboard' && label(other.label) === 'Keep one-sentence journey description' &&
/^Render the accepted journey as a step\/user-does\/user-feels\/plan-specifies table \(\d+ rows covering happy path, save failure, cancel with dirty state, first-time new account\)\. No new design decisions — pure rendering of existing specs\. Completeness: \d+\/10 — implementers understand the emotional arc and can verify the spec covers each moment\.$/.test(description) &&
/^Leave the current one-sentence happy-path description\. Completeness: \d+\/10 — the journey exists but reads like a state machine; error recovery arcs and first-time experience aren't visible without cross-referencing multiple paragraphs\.$/.test(alternative);
}
return false;
}
-41
View File
@@ -1,41 +0,0 @@
/** Accepted interaction behavior surrounding the five seeded visual gaps. */
export const designCountExistingInteractionStates = [
'The existing router protects dirty edits on every in-app exit, including',
'persistent app navigation, using the same Cancel confirmation dialog.',
'Register the browser-native beforeunload warning only while the form is dirty;',
'remove it when clean. Confirmed in-app navigation uses the existing destination',
'heading focus behavior; Keep editing returns focus to the attempted exit.',
'During Save or Export, both request buttons use aria-disabled=true plus an',
'explicit click/keyboard activation guard, rather than the HTML disabled attribute.',
'They remain focusable and keep the existing disabled appearance. Reset and',
'Cancel use HTML disabled during the request. Do not move focus while pending',
'or after success. On a network error, focus the operation-specific Retry only',
'if focus is still on the request trigger; never steal focus the user moved.',
'The existing InlineStatus text stays unchanged while Save is pending:',
'Unsaved changes for a dirty form, otherwise its saved timestamp or initial',
'blank text. Pending feedback belongs to the request button; do not repeat',
'Saving… in the status live region. Success and failure use the outcomes above.',
'When clean and idle, Reset is disabled because it has nothing to discard,',
'and Cancel navigates back immediately without a confirmation. When dirty',
'and idle, Reset and Cancel use their existing discard confirmations. Their',
'44px geometry is unchanged; the disabled style is separate from pending feedback.',
'The existing ErrorSummary mounts in the status/error area below the action',
'group and above Profile. It links each invalid field; focus goes to the first',
'invalid field and the summary is not a second live region. Preserve that slot.',
'The existing error/Retry row is inline above 640px with an 8px gap. At 640px',
'and below, Retry wraps below the text as a full-width 44px ghost button,',
'outside the live region; long errors fit 320px without horizontal scroll.',
// The September 20 retry correctly surfaced these three missing contracts
// in addition to the five seeded visual gaps. They belong to the existing UI.
'The existing operation-specific network error copy is:',
'Save: “Couldn’t save your changes. Your edits are still here.”',
'Export: “Couldn’t prepare your export.” Load: “Couldn’t load your settings.”',
'Each uses the existing error icon and its sibling Retry with the operation-specific',
'accessible label already specified. Preserve edits and the existing retry behavior.',
'Save stays enabled and focusable while idle, whether clean or dirty.',
'A clean Save is a no-op: no request, validation, pending state, timestamp, status, or focus change.',
'Only a dirty Save sends the existing atomic request.',
'The existing Export filename is account-settings-YYYY-MM-DD.json, using the user’s local calendar date',
'at export activation and no account identifiers, including no account name or email. Repeated same-day exports keep the browser’s normal collision suffix',
'(for example, “ (1)”); the application does not overwrite an earlier download.',
];
-42
View File
@@ -1,42 +0,0 @@
import type { AskUserQuestionFingerprint } from './claude-pty-runner';
/** Seeded-count fixtures cover native review cadence; outside voices have separate evals. */
export function pickDesignCountOutsideVoices(
_routing: AskUserQuestionFingerprint,
active: AskUserQuestionFingerprint,
): number | null {
const call = active.nativeCall;
let question: string;
let labels: string[];
if (call) {
if (call.answered || call.failed) return null;
const index = active.nativeQuestionIndex ?? (call.questions.length === 1 ? 0 : undefined);
if (index === undefined || !Number.isInteger(index) || index < 0 || index >= call.questions.length) return null;
const identity = `${call.sessionId}:${call.toolUseId}` +
(call.questions.length > 1 ? `:question:${index}` : '');
if (active.signature !== identity) return null;
const q = call.questions[index]!;
if (q.multiSelect || !/^outside(?: design)? voices$/i.test(q.header.trim()) ||
!/<gstack-qid:(?:outside-voices-design|plan-design-review-outside-voices)>/.test(q.question)) return null;
question = q.question;
labels = q.options.map(option => option.label);
} else {
// Native JSONL can arrive after the answer. The caller supplies the
// active viewport fingerprint; a known but unmatched packet is blocked
// before this hook. Require the specific opt-in premise and both actions.
question = active.promptSnippet;
const packetBar = /^←[^→]*[☐☒]\s+Outside voices\b[^→]*✔\s*Submit\s*→\s*[│┃]?\s*/i.exec(question);
if (packetBar) question = question.slice(packetBar[0].length);
else if (!/^(?:[☐□]\s*)?outside(?: design)? voices\b/i.test(question)) return null;
labels = active.options.map(option => option.label);
while (labels.length > 2 && /^(?:Type something\.?|Chat about this)$/i.test(labels.at(-1)!.trim())) labels.pop();
}
if (!/\b(?:want|run|include|enable)\b[^?]{0,90}\boutside(?: design)? voices\b/i.test(question) ||
!/\b(?:before|for)\s+(?:the\s+)?(?:detailed\s+)?(?:design\s+)?review\b/i.test(question)) return null;
labels = labels.map(label => label.trim().replace(/\s*\(recommended\)\s*$/i, ''));
if (labels.length !== 2) return null;
const yes = labels.map(label => /^Yes,?\s+run outside(?: design)? voices$/i.test(label));
const no = labels.map(label => /^No,?\s+proceed without$/i.test(label));
if (yes.filter(Boolean).length !== 1 || no.filter(Boolean).length !== 1) return null;
return no.findIndex(Boolean) + 1;
}
-970
View File
@@ -1,970 +0,0 @@
import { designFirstReviewAUQ, designReviewSetupAUQ } from './claude-pty-runner';
import type { AskUserQuestionFingerprint } from './claude-pty-runner';
import { pickDesignCountOutsideVoices } from './design-count-outside';
// Native header / question-ID vocabulary this review assigns only to setup and navigation.
const DESIGN_SETUP_HEADER = /^(?:focus|scope|learnings|routing|next steps?|outside(?: design)? voices)$/i;
const DESIGN_SETUP_ID = /(?:^|-)(?:focus|scope|setup|routing|learnings|onboarding|next-steps?|posture|mockups?|target)(?:-|$)/i;
const questionId = (question: string) => /<gstack-qid:\s*([a-z0-9-]+)\s*>/i.exec(question)?.[1] ?? '';
/** Choosing reviewer participation is setup, even when numbered or asked late. */
export function isDesignCountSetup(fp: AskUserQuestionFingerprint): boolean {
if (designReviewSetupAUQ(fp)) return true;
const call = fp.nativeCall;
if (!call?.answered || call.failed || call.questions.length !== 1 ||
call.unansweredQuestionIndices?.length || fp.signature !== `${call.sessionId}:${call.toolUseId}`) return false;
const q = call.questions[0]!;
if (q.multiSelect || !/^outside(?: design)? voices$/i.test(q.header.trim()) ||
(q.question.match(/<gstack-qid:/g)?.length ?? 0) !== 1 ||
!/<gstack-qid:(?:plan-design-review-outside-voices|outside-voices-design)>\s*$/.test(q.question) ||
(q.question.match(/\?/g)?.length ?? 0) !== 1 ||
!/^(?:D\s*\d+(?:\s*\(Step\s*0[A-Z]?\))?\s*[—–:-]\s*)?(?:Run|Want|Include|Enable)\s+outside(?: design)? voices\s+(?:before|for)\s+the\s+(?:detailed\s+)?(?:design\s+)?review(?:\s+passes)?\?/i.test(q.question.trim())) return false;
const labels = q.options.map(option => option.label.trim().replace(/\s*\(recommended\)\s*$/i, ''));
// Consume the entire menu, not just its opening question or action labels.
// Unknown explanatory prose can contain a second product decision.
const remainder = q.question.slice(q.question.indexOf('?') + 1).replace(/<gstack-qid:[^>]+>\s*$/, '').trim();
if (remainder && !/^(?:Codex evaluates the design; a Claude subagent reviews completeness\.|Codex evaluates against OpenAI's design hard rules \+ litmus checks; a Claude subagent does an independent completeness review\. \(Requires Codex CLI to be installed\.\))$/.test(remainder)) return false;
const descriptions = q.options.map(option => (option.description ?? '').trim().replace(/\s+/g, ' '));
const noDescription = /^(?:Skip Codex \+ Claude subagent outside pass\. Best for this case: it's a scoped settings form update with a complete DESIGN\.md; hard-rejection checks apply to marketing surfaces, not OPERATE\/settings UI\.|Skip outside voices and go straight to the 7 review passes\. Faster; sufficient for most plans\.)$/;
const yesDescription = /^(?:Run Codex against OpenAI design hard rules \+ litmus checks, and a separate Claude subagent for an independent completeness review\. Adds time but catches anything a single-model pass misses\.|Launches Codex design critique \+ Claude subagent completeness review in parallel before the 7 passes\. Adds 1[–-]2 minutes\.)$/;
if (labels.some((label, index) => descriptions[index] &&
!(/^No\b/.test(label) ? noDescription : yesDescription).test(descriptions[index]!))) return false;
const no = labels.filter(label => /^No(?:\s*[,—–-]\s*|\s+)proceed without$/i.test(label));
const yes = labels.filter(label => /^Yes(?:\s*[,—–-]\s*|\s+)run (?:outside(?: design)? voices|Codex \+ Claude subagent)$/i.test(label));
return labels.length === 2 && no.length === 1 && yes.length === 1 &&
q.options.some(option => option.label === call.answers?.[q.question]);
}
/** A numbered design-system amendment can be the first review decision. */
function numberedVisualHierarchyFinding(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call || call.answered !== true || call.failed !== false || !call.sessionId || !call.toolUseId ||
call.questions.length !== 1 || !Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length ||
fp.signature !== `${call.sessionId}:${call.toolUseId}` ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0)) return false;
const q = call.questions[0]!;
const finding = /^Gap ([1-9]\d*) of ([1-9]\d*)\s*[—–-]\s*([A-Za-z][A-Za-z0-9_-]{0,39}) button visual hierarchy: apply DESIGN\.md primary button style\?$/i.exec(q.question.trim());
if (!finding || Number(finding[1]) > Number(finding[2]) ||
!new RegExp(`^Gap ${finding[1]}: Button$`, 'i').test(q.header.trim()) || q.multiSelect || q.options.length !== 2 ||
fp.options.length !== 2 || !fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label) ||
!q.options.some(o => o.label === call.answers?.[q.question])) return false;
const labels = q.options.map(o => o.label.trim().replace(/\s*\(recommended\)\s*$/i, ''));
const apply = labels.findIndex(s => /^Apply DESIGN\.md fix$/i.test(s));
const defer = labels.findIndex(s => /^Defer to implementation$/i.test(s));
if (apply < 0 || defer < 0 || apply === defer) return false;
const control = '[A-Za-z][A-Za-z0-9_-]{0,39}';
const amendment = new RegExp(`^Add to plan: ${finding[3]} gets #[0-9a-f]{6} filled \\+ (?:white|black) text \\(primary\\); ${control}(?:, ${control})*(?:,? and ${control})? get neutral ghost style\\. Closes the visual hierarchy gap exactly as DESIGN\\.md specifies\\. Implementation task T[1-9]\\d* becomes committed\\.$`, 'i');
// Both offered bodies describe the actual style amendment or its deferral;
// readiness, a source-selection question, or an example is not this finding.
return amendment.test(q.options[apply]!.description?.trim() ?? '') &&
/^Leave the gap named but unresolved\. Engineer decides the button styles at implementation time without a spec\. Risk: inconsistency with the design system or re-work after review\.$/i.test(q.options[defer]!.description?.trim() ?? '');
}
interface PrimaryFindingFacts {
primary: string;
currentGap: boolean;
controlCount: number;
namedPeers?: string[];
remedy: { peers: string[]; role: boolean; tokens: boolean; authority: boolean; current: boolean };
alternative: { unresolved: boolean; retainedCounts: number[]; current: boolean };
}
/** Presentation adapters supply facts; this is the shared finding boundary. */
function validPrimaryFinding(facts: PrimaryFindingFacts): boolean {
const peers = facts.remedy.peers;
return facts.currentGap && facts.controlCount > 1 && peers.length + 1 === facts.controlCount &&
peers.every(peer => /^[a-z][a-z0-9 _-]{0,39}$/i.test(peer)) &&
new Set(peers).size === peers.length && !peers.includes(facts.primary.toLowerCase()) &&
(!facts.namedPeers || JSON.stringify(peers) === JSON.stringify(facts.namedPeers)) &&
facts.remedy.role && facts.remedy.tokens && facts.remedy.authority && facts.remedy.current &&
facts.alternative.unresolved && facts.alternative.current &&
facts.alternative.retainedCounts.every(count => count === facts.controlCount);
}
/** A qidless Issue with its own design gap is a finding, independent of D numbering. */
function ordinaryDesignIssue(fp: AskUserQuestionFingerprint, scope: { primary: boolean; ownsPrimaryPremise?: boolean }): boolean {
const call = fp.nativeCall;
if (!call?.questions.length) return false;
const nativeValid = call.answered === true && call.failed === false && !!call.sessionId && !!call.toolUseId &&
call.questions.length === 1 && Array.isArray(call.unansweredQuestionIndices) && !call.unansweredQuestionIndices.length &&
fp.signature === `${call.sessionId}:${call.toolUseId}` &&
(fp.nativeQuestionIndex === undefined || fp.nativeQuestionIndex === 0);
const q = call.questions[0]!;
const title = q.question.split('\n')[0]!.trim();
const headerIdentity = /^Issue ([1-9]\d*)(?:: ([A-Za-z][A-Za-z0-9 _-]{0,39}))?$/i.exec(q.header.trim());
const titleSubject = title.replace(/^D[1-9]\d*\s*[—–:-]\s*/i, '')
.replace(/^Issue [1-9]\d*: /i, '');
// Identity can live in either native field. A local actor/role relation is
// distinct from the surrounding decision wording; both fields must agree
// when they name an issue or actor. The facts below still prove the finding.
const actorRole = /^(?!(?:How|What|Which|Who|Why|Where|When)\b)([A-Za-z][A-Za-z0-9 _-]{0,39}) (?:become|be|is|as) (?:the )?(?:(?:only|single|visible|visually) )?(?:filled )?primary (?:header )?action\b/i;
const roleSubject = (text: string) => text.replace(/^(?:Should|Can|Could|Must|Will|Would) /i, '');
const titleRole = actorRole.exec(roleSubject(titleSubject));
// Recognizing this actor/role family is separate from accepting evidence.
// Quotation or invalid native identity must not reopen generic qid fallback.
scope.primary = !!titleRole || actorRole.test(roleSubject(titleSubject.replace(/^["“'‘—–\s]+|["”'’\s]+$/g, '')));
const headerOwnedIssue = headerIdentity && titleRole && !/\bIssue [1-9]\d*\b/i.test(titleSubject) &&
!/\b(?:reviewer|scope|setup|routing|learnings|outside voices|next steps?)\b/i.test(titleSubject)
? [title, headerIdentity[1]!, titleSubject] : null;
// This primary-action decision can name the control in its Issue header.
// F labels annotate findings; they do not establish review identity alone.
const headerActionIssue = /^(?:D[1-9]\d*\s*[—–:-]\s*)?Issue ([1-9]\d*)(?: \(F[1-9]\d*\))?: (How should the header action group establish the primary action)\?$/i.exec(title);
const signaledPrimaryIssue = /^(?:D[1-9]\d*\s*[—–:-]\s*)?Issue ([1-9]\d*) \(G[1-9]\d*\): (How should the header action group signal that [A-Za-z][A-Za-z0-9 _-]{0,39} is the primary action)\?$/i.exec(title);
// Parse the owned comparison independently of the following decision's prose.
// Both forms return the same issue, primary and peer facts for the checks below.
const comparisonTitle = /^(?:D[1-9]\d*\s*[—–:-]\s*)?Issue ([1-9]\d*): ([A-Za-z][A-Za-z0-9 _-]{0,39}) (?:is visually identical to|is indistinguishable from) ([A-Za-z][A-Za-z0-9 ,/_-]{0,119}?)(?: in the header)?\. ([^?]+\?)$/i.exec(title);
const compoundPrimaryIssue = comparisonTitle && /\b(?:fix|resolve|distinguish(?:ed)?|primary action)\b/i.test(comparisonTitle[4]!) ? comparisonTitle : null;
const distinguishedPrimaryIssue = /^D[1-9]\d*\s*[—–:-]\s*Issue ([1-9]\d*): How should ([A-Za-z][A-Za-z0-9 _-]{0,39}) (?:be distinguished|stand out) from ([A-Za-z][A-Za-z0-9 ,/_-]{0,119}?)(?: in the header)?\?$/i.exec(title) ?? compoundPrimaryIssue;
const questionIssue = /^(?:D[1-9]\d*\s*[—–:-]\s*)?Issue ([1-9]\d*)(?: \((?:(?:G[1-9]\d*|Pass [1-7]), )?(?:Visual Hierarchy|Spacing|Color|Typography|Motion)\))?: ([^?]+)\?$/i.exec(title) ?? headerActionIssue ?? signaledPrimaryIssue;
// A declaration can own the same primary-action decision. Its body and
// native choices below must prove the gap, complete styling and deferral.
const declaredPrimaryIssue = (!questionIssue || /\nELI10: (?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) header buttons currently /i.test(q.question)) &&
/^(?:D[1-9]\d*\s*[—–:-]\s*)?Issue ([1-9]\d*)(?: \(G[1-9]\d*\))?: ([^?\n]+)\??$/i.exec(title) ||
(titleRole && (questionIssue ?? headerOwnedIssue));
const issue = questionIssue || declaredPrimaryIssue;
if (!issue) return false;
const subject = issue[2]!.replace(/\.$/, '');
const declaredGap = declaredPrimaryIssue && /\(G([1-9]\d*)\)/.exec(title)?.[1];
const descriptivePrimaryHeader = (signaledPrimaryIssue && /^(?!(?:focus|scope|setup|routing|learnings|outside voices|next steps?)$)[A-Za-z][A-Za-z _-]{0,39}$/i.test(q.header.trim())) ||
(distinguishedPrimaryIssue && /^(?:Visual )?Hierarchy$/i.test(q.header.trim())) ||
(compoundPrimaryIssue && (q.header.trim().toLowerCase() === `${compoundPrimaryIssue[2]} primary`.toLowerCase() ||
q.header.trim().toLowerCase() === `Issue ${compoundPrimaryIssue[1]} ${compoundPrimaryIssue[2]}`.toLowerCase()));
// The numbered headline must ask about a concrete design requirement.
// Reviewer participation or workflow navigation can also use Issue labels.
if (!distinguishedPrimaryIssue && !/\b(?:buttons?|primary(?: header)? actions?|primary emphasis|primacy|hierarchy|spacing|contrast|colou?rs?|labels?|typography|fonts?|loading|spinner|skeleton|motion)\b/i.test(issue[2]!)) return false;
const choiceLabel = (label: string) => label.trim().replace(/^[1-9]\d*[A-Z](?:\s*[—–).:]\s*|\s+)/, '').replace(/\s*\(recommended\)\s*$/i, '');
const opposed = q.options.filter(o => /^(?:Defer|Decline|Leave|Keep|Accept the gap)\b/i.test(choiceLabel(o.label)) ||
(distinguishedPrimaryIssue && /^(?:Keep|Leave)\b/i.test(o.description?.trim() ?? '')));
const repair = !headerActionIssue && !distinguishedPrimaryIssue && !declaredPrimaryIssue && /\b(?:fix|resolve|address)\b/i.test(title) &&
q.options.some(o => /\b(?:closing|closes|fixes|resolves?|applies?)\b/i.test(o.description ?? ''));
// A source citation alone can describe a report or the next reviewer.
// Bind the alternate wording to a named control's concrete style amendment
// and the opposed choice that leaves the documented violation unresolved.
const primary = /^Make ([A-Za-z][A-Za-z0-9 _-]{0,39}) the (?:visible|visually|(?:only|single)(?: filled| visually)?) primary (?:header )?action(?: in the header)?$/i.exec(subject) ??
/^([A-Za-z][A-Za-z0-9 _-]{0,39}) (?:has no|lacks) (?:visual primacy|primary emphasis)(?: in the header action group)?$/i.exec(subject) ??
/^Give ([A-Za-z][A-Za-z0-9 _-]{0,39}) primary emphasis in the header action group$/i.exec(subject) ??
/^How should the header action group signal that ([A-Za-z][A-Za-z0-9 _-]{0,39}) is the primary action$/i.exec(subject) ??
/^How should (?:the )?(?:header )?actions establish that ([A-Za-z][A-Za-z0-9 _-]{0,39}) is the primary action$/i.exec(subject) ??
titleRole ??
(distinguishedPrimaryIssue && [distinguishedPrimaryIssue[0], distinguishedPrimaryIssue[2]!]) ??
(headerActionIssue && new RegExp(`^Issue ${issue[1]}: ([A-Za-z][A-Za-z0-9 _-]{0,39})$`, 'i').exec(q.header.trim()));
scope.primary ||= !!primary;
// Recognize the ordinary parser's premise grammar before validating its
// evidence. Invalid source/roles/status within that grammar must not fall
// through to the broader native-field parser merely by adding gap prose.
const rawAssessments = [...q.question.matchAll(/^(?:>\s*)?ELI10: (.+)$/gm)].map(match => match[1]!);
scope.ownsPrimaryPremise = !!primary && (!!titleRole || !!declaredPrimaryIssue || !!compoundPrimaryIssue || rawAssessments.some(value =>
/^(?:(?:Right now|Today) )?[A-Za-z][A-Za-z0-9 ,/_-]{0,159}? (?:(?:all )?look (?:the same|identical)|are (?:all )?(?:(?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) )?identical buttons)\b/i.test(value) &&
!/^(?:Right now|Today) (?:all|the)\b|\bcurrently\b/i.test(value) ||
/^(?:Right now|Today) (?:all|the) (?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) header buttons (?:look (?:the same|identical)|(?:are|have|share) the same [A-Za-z ,]+)\./i.test(value)));
if (!nativeValid) return false;
const ownedPrimaryHeader = primary && (q.header.trim().toLowerCase() === `${primary[1]} primary`.toLowerCase() ||
q.header.trim().toLowerCase() === `Issue ${issue[1]} ${primary[1]}`.toLowerCase());
if (!(new RegExp(`^Issue ${issue[1]}(?:: [A-Za-z][A-Za-z0-9 _-]{0,39})?$`, 'i').test(q.header.trim()) || descriptivePrimaryHeader || ownedPrimaryHeader) ||
/<gstack-qid:/i.test(q.question) || q.multiSelect ||
q.options.length < 2 || new Set(q.options.map(o => o.label)).size !== q.options.length ||
fp.options.length !== q.options.length || !fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label) ||
!q.options.some(o => o.label === call.answers?.[q.question])) return false;
if (declaredPrimaryIssue && (!primary || q.options.length > 4 || Object.keys(call.answers ?? {}).length !== 1)) return false;
// An attributed native option can put the fill before or after its color.
// It still names the primary, every ghost peer and DESIGN.md in one action.
const namedTokenStyle = primary && new RegExp(`^(?:✅\\s*)?(?:Matches DESIGN\\.md exactly|(?:Apply|Use) DESIGN\\.md(?: tokens)?): ${primary[1]} ` +
'(?:#[0-9a-f]{6} filled(?: with)? (?:white|black) text|filled #[0-9a-f]{6}(?: with)? (?:white|black) text); ' +
'([A-Za-z][A-Za-z0-9 ,/_-]{0,99}) (?:as )?neutral ghost(?: buttons)?\\.', 'i');
const namedTokenIssue = !!namedTokenStyle && q.options.some(o => namedTokenStyle.test(o.description ?? ''));
const primaryEmphasisIssue = namedTokenIssue || !!signaledPrimaryIssue || !!distinguishedPrimaryIssue || !!declaredPrimaryIssue || /^Give [A-Za-z][A-Za-z0-9 _-]{0,39} primary emphasis in the header action group$/i.test(issue[2]!);
const scopedPrimaryStatus = !!headerActionIssue || primaryEmphasisIssue;
const explicitStyle = primary && `${primary[1]} filled (?:primary )?#[0-9a-f]{6}(?:/| with )(?:white|black)(?: text)?; ` +
'[A-Za-z][A-Za-z0-9 ,/_-]{0,99} neutral ghost(?: buttons)?\\.';
const amendments = primary && [
new RegExp(`^(?:✅\\s*)?Matches DESIGN\\.md exactly: ${primary[1]} filled #[0-9a-f]{6} with (?:white|black) text; ` +
'[A-Za-z][A-Za-z0-9 ,_-]{0,99} as neutral ghost buttons\\.', 'i'),
new RegExp(`^(?:✅\\s*)?${primary[1]} becomes the (?:single|one|only) filled(?: primary)?(?: button)? \\(#[0-9a-f]{6}, (?:white|black) text\\); ` +
'[A-Za-z][A-Za-z0-9 /,_-]{0,99} (?:become|are) neutral ghost buttons (?:exactly as DESIGN\\.md specifies|per DESIGN\\.md)\\b', 'i'),
new RegExp(`^(?:✅\\s*)?${primary[1]} is the (?:single|only) filled #[0-9a-f]{6} button; ` +
'[A-Za-z][A-Za-z0-9 /,_-]{0,99} become neutral ghosts, exactly (?:per DESIGN\\.md|as DESIGN\\.md prescribes)\\b', 'i'),
new RegExp(`^(?:✅\\s*)?Apply DESIGN\\.md(?: tokens)?: ${primary[1]} #[0-9a-f]{6} filled(?: with)? (?:white|black) text; ` +
'[A-Za-z][A-Za-z0-9 ,/_-]{0,99} neutral ghost(?: buttons)?\\.', 'i'),
// The same concrete style can cite DESIGN.md before or after its tokens.
new RegExp(`^(?:✅\\s*)?(?:Apply DESIGN\\.md(?: tokens)?: ${explicitStyle}|${explicitStyle} Exact DESIGN\\.md\\.)`, 'i'),
new RegExp(`^(?:✅\\s*)?${primary[1]}\\s*=\\s*filled #[0-9a-f]{6} with (?:white|black) text; ` +
'[A-Za-z][A-Za-z0-9 ,/_-]{0,99}\\s*=\\s*neutral ghost(?: buttons?)?,? per DESIGN\\.md\\b', 'i'),
];
const primaryHeader = !q.header.includes(':') || q.header.split(':')[1]!.trim().toLowerCase() === primary?.[1]?.toLowerCase();
const ownedStatus = (value: string, index: number, source: string) =>
/^(?:withdrawn|superseded|resolved|closed|historical|hypothetical|rejected|cancelled|canceled|not current|no longer current)$/i.test(value) &&
/(?:^|[.!?;]\s+|\n)(?:[✅❌]\s*)?(?:Correction:\s*)?(?:(?:(?:This|That|The) (?:issue|finding|question|amendment|deferral|style|fix|remedy|choice|option|(?:DESIGN\.md |token )?(?:requirement|contract))|(?:Issue |G)[1-9]\d*) (?:is|was|has been)|(?:these|the|this) (?:tokens?|styles?|primary treatment) (?:are|is|were|was|have been|has been)) $/i.test(source.slice(0, index));
// The style wordings share one owned decision: a current equal-weight gap,
// a named control's DESIGN.md amendment, and a different choice retaining it.
// A following status assertion remains current after a parenthesized effort
// estimate. Preserve the estimate and expose its boundary to the same guards.
const currentText = (text: string) => (scopedPrimaryStatus
? text.replace(/(\(human: ~?[0-9]+(?:\.[0-9]+)?(?:h|min) \/ CC: ~?[0-9]+(?:\.[0-9]+)?(?:h|min)\))(?=\s+\S)/g, '$1.')
.replace(/\(recommended\)(?=\s+\S)/gi, '$&.')
: text)
.replace(/```[\s\S]*?(?:```|$)|~~~[\s\S]*?(?:~~~|$)/g, '')
.replace(/^(?:\s*>| {4}|\t).*$/gm, '')
.replace(/`([^`]+)`/g, (_, body: string, index: number, source: string) =>
scopedPrimaryStatus && ownedStatus(body, index, source) ? body : /\s/.test(body) ? '' : body)
// A quoted status scalar remains a current assertion when its unquoted
// subject names this decision; whole quoted historical prose stays absent.
.replace(scopedPrimaryStatus
? /"[^"\n]*"|“[^”\n]*”|(?<!\w)'[^'\n]*'(?!\w)|‘[^’\n]*’/g
: /"[^"\n]*"|“[^”\n]*”/g, (quoted: string, index: number, source: string) =>
ownedStatus(quoted.slice(1, -1), index, source)
? quoted.slice(1, -1) : '').replace(/\*\*/g, '');
const questionText = currentText(q.question);
const assessments = [...questionText.matchAll(/^ELI10: (.+)$/gm)];
const prefix = questionText.slice(0, assessments[0]?.index ?? 0)
.split('\n').filter(line => line.trim()).slice(1);
const sourceAssessment = /\b(?:historical|hypothetical|quoted|source|earlier review)\s+(?:example|excerpt|assessment|material|text)\b|\bnot\s+(?:the\s+)?current\s+(?:UI|assessment|finding|amendment|deferral|remedy|choice|option)\b|\bthis (?:finding|amendment|deferral|remedy|choice|option) (?:applies only to|belongs to) (?:an? )?(?:another|different) (?:project|plan|review)\b/i;
const assessment = assessments.length === 1 &&
prefix.every(line => /^(?:Project\/branch\/task:|\[P[0-3]\])/.test(line)) &&
!/^(?:Project\/branch\/task:|\[P[0-3]\])\s*(?:If|When|Unless|Provided|Assuming)\b/im.test(prefix.join('\n')) &&
!sourceAssessment.test(prefix.join(' ')) && !sourceAssessment.test(assessments[0]![1]!)
? assessments[0]![1]! : '';
const headerPeers = primary && headerActionIssue && new RegExp(`^${primary[1]}, ([A-Za-z][A-Za-z0-9 _-]{0,39}(?:, [A-Za-z][A-Za-z0-9 _-]{0,39})*(?:,? and [A-Za-z][A-Za-z0-9 _-]{0,39})?) currently look (?:the same|identical)\\.`, 'i').exec(assessment);
// Equal visual properties can establish the same current lack of hierarchy.
// Shared geometry alone is not a claim that the actions look equally primary.
const properties = '(?:size|weight|colou?r|fill|emphasis)(?:(?:, ?|,? and )(?:size|weight|colou?r|fill|emphasis))*';
const equalProperties = distinguishedPrimaryIssue && new RegExp('^(?:Right now|Today) (?:all|the) (two|three|four|five|six|seven|eight|nine|ten|[1-9]\\d*) header buttons (?:are|have|share) the same (' + properties + ')\\.', 'i').exec(assessment);
const countedHeader = equalProperties && /\b(?:weight|colou?r|fill|emphasis)\b/i.test(equalProperties[2]!) && equalProperties ||
distinguishedPrimaryIssue && /^(?:Right now|Today) (?:all|the) (two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) header buttons look (?:the same|identical)\./i.exec(assessment) ||
distinguishedPrimaryIssue && /^The header shows (two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) buttons that look exactly alike\./i.exec(assessment) ||
(distinguishedPrimaryIssue || declaredPrimaryIssue) && /^(two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) buttons (?:sit in a row and |in a row )all look the same[,.]/i.exec(assessment) ||
declaredPrimaryIssue && /^(two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) header buttons currently (?:share one style|look identical)\./i.exec(assessment);
const primaryAssessment = distinguishedPrimaryIssue || declaredPrimaryIssue ? countedHeader?.[0] : headerActionIssue ? headerPeers?.[0] :
primary && new RegExp(`^(?:Right now|Today) ${primary[1]}(?:, [A-Za-z][A-Za-z0-9 _-]{0,39})+(?:,? and [A-Za-z][A-Za-z0-9 _-]{0,39})? (?:(?:all )?look (?:the same|identical)|are (?:all )?(?:(?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\\d*) )?identical buttons)\\b`, 'i').exec(assessment)?.[0];
const premiseSentence = assessment.split(/[.!?](?:\s|$)/)[0] ?? '';
const currentPremise =
!/\b(?:archived|historical|hypothetical|quoted|example|previous|earlier)\b/i.test(premiseSentence) &&
!/\bPLAN\.md (?:onboarding|post-review TODO|engineering review)\b/i.test(prefix.join(' '));
const currentPrimary = !!primaryAssessment && !/\b(?:not|never|no longer)\b/i.test(primaryAssessment) && currentPremise;
// The current assessment can state the full token contract while an offered
// amendment names the existing component variants that implement it.
const numberValue = (value: string) => /^\d+$/.test(value) ? Number(value) :
['zero', 'one', 'two', 'three', 'four', 'five', 'six', 'seven', 'eight', 'nine', 'ten'].indexOf(value.toLowerCase());
const controlNames = (text: string) => text.toLowerCase().split(/\s*\/\s*|,\s*(?:and\s+)?|\s+and\s+/).map(s => s.trim()).sort();
const validControls = (controls: string[]) => controls.length > 0 &&
controls.every(control => /^[a-z][a-z0-9 _-]{0,39}$/i.test(control)) && new Set(controls).size === controls.length;
const headerControls = distinguishedPrimaryIssue ? controlNames(distinguishedPrimaryIssue[3]!) : headerPeers ? controlNames(headerPeers[1]!) : [];
const namedPremiseMatch = /^(?:(?:Right now|Today) )?([A-Za-z][A-Za-z0-9 ,/_-]{0,159}?) (?:(?:all )?look (?:the same|identical)|are (?:all )?((?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) )?identical buttons)\b/i.exec(assessment);
const namedPremise = primary && namedPremiseMatch &&
new RegExp(`\\b${primary[1]}\\b`, 'i').test(namedPremiseMatch[1]!) ? namedPremiseMatch : null;
const premiseActors = namedPremise ? controlNames(namedPremise[1]!) : [];
const namedCurrentGap = primary && namedPremise && validControls(premiseActors) &&
premiseActors.includes(primary[1]!.toLowerCase()) && premiseActors.length > 1 &&
currentPremise && !/\b(?:not|never|no longer)\b/i.test(namedPremise[0]) &&
(!namedPremise[2] || numberValue(namedPremise[2].trim()) === premiseActors.length);
const premiseCount = countedHeader ? numberValue(countedHeader[1]!) : namedCurrentGap ? premiseActors.length : 0;
const otherControls = headerActionIssue || distinguishedPrimaryIssue ? headerControls.length : primary && primaryAssessment
? primaryAssessment.replace(new RegExp(`^(?:Right now|Today) ${primary[1]},\\s*`, 'i'), '')
.replace(/\s+(?:(?:all )?look (?:the same|identical)|are (?:all )?(?:(?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) )?identical buttons)$/, '')
.split(/,\s*(?:and\s+)?|\s+and\s+/).length : 0;
const variantContract = primary && new RegExp(`(?:^|[.!?]\\s+)DESIGN\\.md already says ${primary[1]} is the only filled button ` +
'\\(#[0-9a-f]{6} with (?:white|black) text(?:, about [0-9]+(?:\\.[0-9]+)?:1 contrast)?\\) and the other ' +
'(two|three|four|five|six|seven|eight|nine|ten|[1-9]\\d*) are neutral ghost buttons\\.', 'i').exec(assessment);
const statusBoundary = scopedPrimaryStatus ? '[.!?;]' : '[.!?]';
const invalidContract = new RegExp(`(?:^|${statusBoundary}\\s+|\\n)(?:[✅❌]\\s*)?(?:Correction:\\s*)?(?:this|that|the) (?:(?:DESIGN\\.md|token) )?(?:requirement|contract) (?:is|was|has been) (?:withdrawn|superseded|rejected|cancelled|canceled|not current|no longer current)\\b`, 'i');
const namedContract = primary && new RegExp(`(?:^|[.!?]\\s+)DESIGN\\.md already says ${primary[1]} is the only filled primary button and the other (two|three|four|five|six|seven|eight|nine|ten|[1-9]\\d*) are neutral ghost buttons\\.`, 'i').exec(assessment);
const headerContract = primary && headerActionIssue && new RegExp(`(?:^|[.!?]\\s+)DESIGN\\.md already answers it: ${primary[1]} is the only filled primary button, the other (two|three|four|five|six|seven|eight|nine|ten|[1-9]\\d*) are neutral ghost buttons\\.`, 'i').exec(assessment);
const conditionalHeader = (text: string) => /(?:^|[.!?;]\s+|\n)(?:[✅❌]\s*)?(?:Correction:\s*)?(?:If|When|Unless|Assuming|Provided)\b/i.test(text) || /\b(?:only if|unless|pending approval|subject to approval)\b/i.test(text);
// Approval conditions suspend this offered decision; explanatory conditions
// about user behavior do not make an otherwise current amendment optional.
const pendingPrimaryApproval = (text: string) => primaryEmphasisIssue && (
/(?:^|[.!?;]\s+|\n)(?:[✅❌]\s*)?(?:Correction:\s*)?(?:If|When|Once|Provided|Assuming|Pending)\s+(?:approval|approved|acceptance|accepted|(?:we|you)\s+(?:approve|accept))\b/i.test(text) ||
(declaredPrimaryIssue && new RegExp(`(?:^|[.!?;]\\s+|\\n)(?:Correction:\\s*)?(?:This (?:issue|finding|amendment|deferral|option)|Issue ${issue[1]}${declaredGap ? `|G${declaredGap}` : ''}) (?:requires approval|applies only if approved)\\b`, 'i').test(text)));
const currentHeaderContract = declaredPrimaryIssue || distinguishedPrimaryIssue ?
(countedHeader || namedCurrentGap) && !conditionalHeader(questionText) : !headerActionIssue || (headerContract && headerControls.length > 0 &&
validControls(headerControls) && !headerControls.includes(primary![1]!.toLowerCase()) &&
numberValue(headerContract[1]!) === headerControls.length && !conditionalHeader(questionText));
const statedVariant = (variantContract || namedContract) &&
numberValue((variantContract || namedContract)![1]!) === otherControls &&
!/\b(?:proposed|hypothetical|quoted|historical|source)\s+(?:example|contract|requirement)\b/i.test(assessment.slice(0, (variantContract || namedContract)!.index)) &&
!invalidContract.test(questionText);
const withdrawn = new RegExp(`(?:^|${statusBoundary}\\s+|\\n)(?:[✅❌]\\s*)?(?:Correction:\\s*)?(?:(?:(?:This|That|The) (?:issue|finding|question|amendment|deferral|style|fix|remedy|choice|option)|Issue ${issue[1]}${declaredGap ? `|G${declaredGap}` : ''}) (?:is|was|has been) (?:withdrawn|superseded|resolved|closed|historical|hypothetical|rejected|cancelled|canceled|not current|no longer current)|We have (?:resolved|closed|withdrawn) this (?:issue|finding)|No current (?:issue|finding|gap|violation) (?:remains|exists))\\b`, 'i');
const closedGap = /(?:^|[.!?;]\s+|\n)(?:[✅❌]\s*)?(?:Correction:\s*)?(?:this|the|that) (?:(?:design )?debt|gap|violation) (?:is|was|has been) (?:already\s+|now\s+)?(?:resolved|fixed|closed)\b/i;
const cancelledStyle = /(?:^|[.!?;]\s+|\n)(?:[✅❌]\s*)?(?:Correction:\s*)?(?:do not|don't|never|skip|cancel|withdraw)\s+(?:apply|use|add|keep)\s+(?:(?:these|the|this)\s+)?(?:tokens?|styles?|primary treatment)\b/i;
const withdrawnStyles = /(?:^|[.!?;]\s+|\n)(?:[✅❌]\s*)?(?:Correction:\s*)?(?:these|the|this) (?:tokens?|styles?|primary treatment) (?:are|is|were|was|have been|has been) (?:withdrawn|rejected|cancelled|canceled|not current|no longer current)\b/i;
const currentOptionEvidence = (text: string) => !pendingPrimaryApproval(text) && !conditionalHeader(text) &&
!sourceAssessment.test(text) && !withdrawn.test(text) && !closedGap.test(text) &&
!invalidContract.test(text) && !cancelledStyle.test(text) && !withdrawnStyles.test(text);
const contradictsDesign = (text: string) => /(?:^|[.!?;]\s+|\n)(?:[✅❌]\s*)?(?:Correction:\s*)?(?:these|the|this) (?:tokens?|styles?|variants?) (?:(?:do|does) not match DESIGN\.md|(?:are|is|were|was) (?:not approved|unapproved))\b/i.test(text);
const consistentRemedyEvidence = (text: string, owner: string, peers: string[]) => currentOptionEvidence(text) &&
!contradictsDesign(text) &&
!new RegExp(`\\b${owner}(?:\\s*[:=]\\s*|\\s+)(?:(?:is|as|becomes) )?(?:the |a )?(?:neutral )?(?:ghost|outlined|secondary)\\b`, 'i').test(text) &&
!peers.some(peer => new RegExp(`\\b(?:primary(?: button| action)? ${peer}(?=$|[\\s,.;])|${peer}(?:\\s*[:=]\\s*|\\s+)(?:(?:is|as|becomes) )?(?:the |a )?(?:filled(?: primary)?|primary|outlined))\\b`, 'i').test(text));
const choiceIds = q.options.map(o => /^([1-9]\d*)[A-Z](?:\s*[—–).:]\s*|\s+)/.exec(o.label));
const primaryRepair = primaryHeader && amendments &&
(declaredPrimaryIssue || distinguishedPrimaryIssue ? currentPrimary || namedCurrentGap : currentPrimary) && currentHeaderContract &&
prefix.filter(line => /^Project\/branch\/task:/.test(line)).length === 1 &&
!!call.answeredAt && Number.isFinite(Date.parse(call.answeredAt)) &&
choiceIds.every(id => id?.[1] === issue[1]) &&
!pendingPrimaryApproval(questionText) &&
!conditionalHeader(titleSubject) && !sourceAssessment.test(titleSubject) &&
currentText(titleSubject) === titleSubject &&
!sourceAssessment.test(questionText) &&
!withdrawn.test(questionText) && !closedGap.test(questionText) && !withdrawnStyles.test(questionText) && !invalidContract.test(questionText) &&
q.options.some(amendment => {
const body = currentText(amendment.description ?? '');
if (pendingPrimaryApproval(body)) return false;
// Roles and their concrete tokens belong to one native option; a familiar
// label alone cannot supply the style or borrow DESIGN.md from a peer.
const roleLabel = currentText(amendment.label);
if (!currentOptionEvidence(roleLabel)) return false;
const propertyStyle = primary && distinguishedPrimaryIssue && new RegExp(`^(?:✅\\s*)?${primary[1]} (?:is|becomes) the (?:only|single) filled(?: primary)? #[0-9a-f]{6}(?: button)? with (?:white|black) text; ([A-Za-z][A-Za-z0-9 ,/_-]{0,119}) (?:are|become) neutral ghost(?: buttons)?\\.`, 'i').exec(body);
const optionAuthority = /(?:^|[.;]\s+)(?:✅\s*)?(?:Matches DESIGN\.md exactly|Per DESIGN\.md)(?=[:.;,]|$)/i.test(body);
const roleAuthority = new RegExp(`^${issue[1]}[A-Z][).:]?\\s+(?:(?:Apply|Use|Reuse) )?DESIGN\\.md\\b`, 'i').test(roleLabel) ||
optionAuthority;
const roleStyle = propertyStyle && roleAuthority && !conditionalHeader(roleLabel) &&
!/\b(?:not|never|no|if|historical|hypothetical|source|quoted|withdrawn|superseded|cancelled|canceled)\b/i.test(roleLabel) &&
!headerControls.some(peer => new RegExp(`\\b(?:primary(?: button| action)? ${peer}|${peer} (?:as )?(?:the )?(?:filled )?primary)\\b`, 'i').test(roleLabel)) ? propertyStyle : null;
const declaredStyle = primary && declaredPrimaryIssue &&
new RegExp(`^(?:✅\\s*)?${primary[1]} filled #[0-9a-f]{6}(?: with)? (?:white|black)(?: text)?; ([A-Za-z][A-Za-z0-9 ,/_-]{0,119}) neutral ghost(?: buttons)?\\.`, 'i').exec(body);
const headerStyle = primary && headerActionIssue && new RegExp(`^(?:✅\\s*)?${primary[1]} becomes the only filled #[0-9a-f]{6} button with (?:white|black) text; ([A-Za-z][A-Za-z0-9 ,_-]{0,119}) become neutral ghost buttons, exactly as DESIGN\\.md states\\.`, 'i').exec(body);
// A descriptive header still owns a concrete primary and every peer.
// Extract the primary token clause and peer clause independently of their
// separator. Their DESIGN.md authority must be in this same native option.
const primaryClause = primary && (distinguishedPrimaryIssue || declaredPrimaryIssue) && new RegExp(`^(?:✅\\s*)?(?:Matches DESIGN\\.md exactly: )?${primary[1]}(?:\\s*[:=]\\s*|\\s+)` +
'(?:(?:is|becomes) (?:the (?:only|single) )?)?(filled(?: primary)?(?: button)? )?' +
'(?:\\(#[0-9a-f]{6}, (?:white|black) text\\)|#[0-9a-f]{6}(?: button)?(?: with)? (?:white|black)(?: text)?)' +
'(?:,\\s*[1-9]\\d*(?:\\.\\d+)?px)?[.;,]\\s+', 'i').exec(body);
const peerClause = primaryClause && /^([A-Za-z][A-Za-z0-9 ,/_-]{0,119}?)(?:\s*[:=]\s*|\s+)(?:(?:are|become|as) )?neutral ghost(?: buttons)?[.;,]/i.exec(body.slice(primaryClause[0].length));
const semanticLabel = choiceLabel(roleLabel);
const labelledRole = primary && (new RegExp(`^${primary[1]} filled primary(?:,|$)`, 'i').test(semanticLabel) ||
new RegExp(`^Filled (?:primary (?:(?:\\+|and|with) ghosts|${primary[1]})|${primary[1]}, ghost others)$`, 'i').test(semanticLabel));
const designAction = /^(?:Apply|Use|Reuse) DESIGN\.md (?:tokens?|styles)$/i.test(semanticLabel);
const namedDesignRole = /^(?:(?:Apply|Use|Reuse) )?DESIGN\.md (?:primary|tokens?|styles?)\b/i.test(semanticLabel);
const approvedTokens = /(?:^|[.;]\s+)(?:✅\s*)?(?:Uses?|Applies?|Reuses?|Matches?) (?:the )?(?:exact )?approved (?:tokens?|styles?)(?=[.;]|$)/i.test(body);
const clauseAuthority = optionAuthority || designAction || (namedDesignRole && approvedTokens) || (primaryClause && peerClause &&
/^exactly per DESIGN\.md(?:[.;]|\n|$)/i.test(body.slice(primaryClause[0].length + peerClause[0].length).trimStart()));
const clauseRole = !!labelledRole || !!primaryClause?.[1];
const styleLabel = labelledRole || designAction || roleStyle || namedDesignRole ||
(/\b(?:filled|primary)\b/i.test(semanticLabel) && !/\b(?:review|reviewer|prepare|start|next|setup|source|example)\b/i.test(semanticLabel));
const clauseStyle = primaryClause && peerClause && clauseAuthority && clauseRole && styleLabel
? [primaryClause[0] + peerClause[0], peerClause[1]!] : null;
const distinguishedStyle = primary && distinguishedPrimaryIssue && (
clauseStyle ??
new RegExp(`^(?:✅\\s*)?${primary[1]} is #[0-9a-f]{6} with (?:white|black) text; ([A-Za-z][A-Za-z0-9 ,_-]{0,119}) are neutral ghost buttons per DESIGN\\.md\\.`, 'i').exec(body) ??
new RegExp(`^(?:✅\\s*)?${primary[1]} becomes the only filled button \\(#[0-9a-f]{6}, (?:white|black) text\\); ([A-Za-z][A-Za-z0-9 ,/_-]{0,119}) use the existing neutral ghost variant` +
'(?: \\(human: ~?[0-9]+(?:\\.[0-9]+)?(?:h|min) / CC: ~?[0-9]+(?:\\.[0-9]+)?(?:h|min)\\))?\\. (?:✅\\s*)?Matches DESIGN\\.md exactly\\b', 'i').exec(body) ?? roleStyle);
const findingStyle = clauseStyle ?? (declaredPrimaryIssue && designAction ? declaredStyle : null) ?? distinguishedStyle;
const attributedStyle = namedTokenStyle?.exec(body);
let namedTokenValid = false;
if (namedTokenIssue) {
const peers = primaryAssessment?.replace(new RegExp(`^(?:Right now|Today) ${primary![1]},\\s*`, 'i'), '')
.replace(/\s+(?:(?:all )?look (?:the same|identical)|are (?:all )?(?:(?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) )?identical buttons)$/, '');
const sources = (prefix.join(' ') + ' ' + assessment).match(/[\w./-]+\.md\b/g) ?? [];
const sourceRoles = [...assessment.matchAll(new RegExp(`(?:^|[.!?]\\s+)DESIGN\\.md (?:already )?(?:says|states|specifies|requires|defines|names the treatment):? ${primary![1]} is the (?:only|single) filled button ` +
'\\((#[0-9a-f]{6})(?:,| with) (white|black) text(?:, (?:~|about )?[0-9]+(?:\\.[0-9]+)?:1 contrast)?\\)(?:,| and) (?:the )?other ' +
'(two|three|four|five|six|seven|eight|nine|ten|[1-9]\\d*) are neutral ghost(?:s| buttons)\\.', 'gi'))];
const assertedRoles = sourceRoles.length === 1 ? sourceRoles[0] : undefined;
// Additional imperative work cannot borrow this styling decision's ACK.
// Explanatory subjects, negated work and quoted history are not commands.
const extraAction = /(?:^|[.!?;]\s+|\n|[✅❌]\s*|\b(?:and|but|while)\s+)(?:(?:also|then|now|first|next|please)\s+)*(?:approve|add|build|create|implement|replace|remove|delete|deploy|install|configure|rewrite|migrate|launch|fix|repair|resolve)\b/i;
const wrongTokens = !assertedRoles || !attributedStyle ||
numberValue(assertedRoles[3]!) !== otherControls ||
attributedStyle[0].match(/#[0-9a-f]{6}/i)?.[0].toLowerCase() !== assertedRoles[1]!.toLowerCase() ||
attributedStyle[0].match(/\b(?:white|black) text\b/i)?.[0].toLowerCase() !== assertedRoles[2]!.toLowerCase() + ' text' ||
extraAction.test(questionText) || q.options.some(o => extraAction.test(currentText(o.label + '\n' + (o.description ?? ''))));
namedTokenValid = Boolean(!wrongTokens && attributedStyle && peers && q.options.length <= 4 && Object.keys(call.answers ?? {}).length === 1 &&
(questionText.match(/^Project\/branch\/task:/gm)?.length ?? 0) === 1 &&
!conditionalHeader(questionText) && !conditionalHeader(body) &&
sources.includes('DESIGN.md') && sources.every(source => ['PLAN.md', 'DESIGN.md'].includes(source)) &&
JSON.stringify(controlNames(attributedStyle[1]!.replaceAll('/', ','))) === JSON.stringify(controlNames(peers)) &&
new Set(controlNames(peers)).size === otherControls && !invalidContract.test(body));
}
const style = declaredPrimaryIssue || distinguishedPrimaryIssue ? findingStyle?.[0] : headerActionIssue ? headerStyle?.[0] : (namedTokenValid ? attributedStyle?.[0] : undefined) ?? amendments.map(pattern => pattern.exec(body)).find(Boolean)?.[0];
if ((declaredPrimaryIssue || distinguishedPrimaryIssue) &&
/(?:^|[.!?;]\s+|\n)(?:[✅❌]\s*)?(?:Correction:\s*)?(?:the|this) (?:current )?(?:amendment|fix) keeps (?:all )?(?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) (?:header )?buttons identical\b/i.test(body)) return false;
if ((declaredPrimaryIssue || distinguishedPrimaryIssue) && (!findingStyle || conditionalHeader(body) || invalidContract.test(body) ||
!(roleStyle || labelledRole || designAction || clauseStyle))) return false;
if (headerActionIssue && (!headerStyle || conditionalHeader(body) || invalidContract.test(body) ||
JSON.stringify(controlNames(headerStyle[1]!)) !== JSON.stringify(headerControls))) return false;
const variantLine = /^✅\s*Uses the existing Button primary and ghost variants from DESIGN\.md; no new styles\./m.exec(body);
const benefits = variantLine ? body.slice(0, variantLine.index).trim().split('\n').filter(Boolean) : [];
const labelledRoles = /^✅ Matches DESIGN\.md exactly: one filled primary, (two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) neutral ghosts, [1-9]\d*px targets kept\./.exec(body);
const labelledStyle = statedVariant && namedContract && labelledRoles &&
numberValue(labelledRoles[1]!) === otherControls &&
new RegExp(`^[1-9]\\d*[A-Z]\\) ${primary![1]} filled #[0-9a-f]{6}/(?:white|black), others ghost(?: \\(recommended\\))?$`, 'i').test(amendment.label);
const variantRepair = !headerActionIssue && (labelledStyle || (statedVariant && variantContract && variantLine &&
new RegExp(`^[1-9]\\d*[A-Z] Filled ${primary![1]}, ghost others(?: \\(recommended\\))?$`, 'i').test(amendment.label) &&
benefits.every(line => /^✅\s*(?!(?:If|When|Unless|Historical|Hypothetical|Quoted|Source|Example)\b)\S/i.test(line)) &&
!/\b(?:archived|historical|hypothetical|quoted|previous|earlier)\b/i.test(benefits.join(' ')))) &&
!/(?:^|[.!?]\s+|\n)(?:Correction:\s*)?(?:do not|don't|never|skip|cancel|withdraw) (?:apply|use|add|keep) (?:the |these )?(?:Button )?primary and ghost variants\b/i.test(body) &&
!contradictsDesign(body) &&
!/(?:^|[.!?]\s+|\n)(?:Correction:\s*)?(?:the|this) (?:current )?amendment keeps all (?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) buttons identical\b/i.test(body);
// A named primary cannot simultaneously occur in the ghost-control list.
const secondaryStyle = findingStyle?.[1] ?? style?.slice(style.indexOf(';') + 1);
if ((!style && !variantRepair) || (secondaryStyle && new RegExp(`\\b${primary![1]}\\b`, 'i').test(secondaryStyle)) ||
sourceAssessment.test(body) || withdrawn.test(body) || closedGap.test(body) || cancelledStyle.test(body) || withdrawnStyles.test(body) || contradictsDesign(body)) return false;
return opposed.some(defer => {
const declined = currentText(defer.description ?? '');
const declinedLabel = currentText(defer.label);
if (!currentOptionEvidence(declinedLabel)) return false;
if (pendingPrimaryApproval(declined) || (namedTokenIssue && (conditionalHeader(declined) ||
/(?:^|[.!?;]\s+|\n)(?:this|the) (?:option|deferral) (?:(?:now|already|actually) )?(?:fixes|resolves|closes) (?:the |this )?(?:hierarchy |primary-action )?gap\b/i.test(declined)))) return false;
const declaredDeferral = declaredPrimaryIssue &&
/^Defer$/i.test(choiceLabel(declinedLabel)) &&
new RegExp(`^Leave ${declaredGap ? `G${declaredGap}` : `Issue ${issue[1]}`} open and record it as unresolved\\.`, 'i').test(declined) &&
!new RegExp(`(?:^|[.!?;]\\s+|\\n)(?:Correction:\\s*)?(?:do not|don't|never|skip|cancel|withdraw) (?:leave|keep|defer) (?:${declaredGap ? `G${declaredGap}|` : ''}Issue ${issue[1]})\\b`, 'i').test(declined) &&
!conditionalHeader(declined) && !sourceAssessment.test(declined) && !withdrawn.test(declined) &&
!closedGap.test(declined) && !invalidContract.test(declined) && !cancelledStyle.test(declined) && !withdrawnStyles.test(declined);
// Native menus can list current benefits before the gap retained by
// declining. Only consume a complete affirmative pro/con prefix; prose
// framing a source example or a future condition cannot expose an icon.
const headerDeferral = /^(?:Leave|Keep) the header unchanged and record the gap as (?:debt|an open issue)\.\s*/i.exec(declined);
const deferralBody = headerDeferral ? declined.slice(headerDeferral[0].length) : declined;
const cancelledHeaderDeferral = headerDeferral && /(?:^|[.!?;]\s+|\n)(?:Correction:\s*)?(?:do not|don't|never|skip|cancel|withdraw) (?:keep|leave) (?:the )?header unchanged\b/i.test(declined);
const pros = /^(?:✅(?!\s*(?:If|When|Unless|Historical|Hypothetical|Quoted|Source|Example)\b)\s*[^✅❌]+)+❌\s*/i.exec(deferralBody);
const remaining = pros && !sourceAssessment.test(pros[0]) ? deferralBody.slice(pros[0].length) : deferralBody;
const retainedEmphasis = distinguishedPrimaryIssue && (
new RegExp(`^(?:Keep|Leave) identical (?:header )?buttons, bold (?:the )?${primary![1]} (?:text|label)\\. (?:Weak(?: visual)? signal, )?off-token\\.`, 'i').test(remaining) ||
new RegExp(`^Weight alone is a weak signal at a glance and violates DESIGN\\.md, which names ${primary![1]} the only filled action\\. (?:❌\\s*)?Leaves the primary action undiscoverable for scanning users\\.`, 'i').test(remaining));
const retainedHierarchyGap = distinguishedPrimaryIssue && /^(?:❌\s*)?Ships a (?:known|documented) DESIGN\.md violation and the plan['’]s own Visual Hierarchy gap (?:stays|remains) open\./i.test(remaining);
const retainedRoleGap = roleStyle && /^(?:No change[.;]\s*)?(?:the |this )?(?:finding|issue|gap) (?:stays|remains) (?:open|unresolved)\b/i.test(remaining);
const retainedControls = /^(?:Leave|Keep) (?:the |all )?(two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) (?:header )?buttons (?:uniform|identical|equal)(?: for now)?(?:[.;]\s+| and )/i.exec(declined);
const unresolvedDebt = retainedControls && numberValue(retainedControls[1]!) === headerControls.length + 1 &&
/^record (?:it|(?:the|this) gap) as (?:unresolved|open) design debt\./i.test(declined.slice(retainedControls[0].length));
const cancelledRetention = /(?:^|[.!?;]\s+|\n)(?:[✅❌]\s*)?(?:Correction:\s*)?(?:do not|don't|never|skip|cancel|withdraw) (?:keep|leave) (?:the |all )?(?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) (?:header )?buttons (?:uniform|identical|equal)\b/i.test(declined);
const cancelledDebt = /(?:^|[.!?;]\s+|\n)(?:[✅❌]\s*)?(?:Correction:\s*)?(?:do not|don't|never|skip|cancel|withdraw) (?:record|log|track) (?:it|this|(?:the|this) gap) as (?:unresolved|open) design debt\b/i.test(declined);
const retainedLabel = /^(?:Keep|Leave) (?:all |the )?(two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) (?:(?:identical|equal|uniform) (?:header )?buttons|(?:header )?buttons (?:identical|equal|uniform))\b/i.exec(choiceLabel(declinedLabel));
const retainedPrimaryGap = retainedLabel && /^Violates DESIGN\.md\b/i.test(remaining) &&
/\bleaves the primary action (?:indistinguishable|undiscoverable)\b/i.test(remaining);
const retainedNoPrimary = retainedLabel && /^Ships the documented violation; no primary action;/i.test(remaining);
// The opposed option's label and body share ownership too. A retained
// actor count can precede ordinary tradeoffs before the explicit gap.
const retainedViolation = retainedLabel && declined.split(/[.!?;]\s+/).some(clause =>
/^(?:❌\s*)?(?:Leaves?|Ships?|Keeps?|Retains?) (?:a )?(?:known |documented )?DESIGN\.md violation\b/i.test(clause) &&
/\bno primary action\b/i.test(clause));
if (declaredPrimaryIssue || distinguishedPrimaryIssue) return defer !== amendment && validPrimaryFinding({
primary: primary![1]!, currentGap: !!(currentPrimary || namedCurrentGap) && !!currentHeaderContract &&
(!namedPremise || !!namedCurrentGap) &&
(!namedCurrentGap || !countedHeader || premiseActors.length === premiseCount) &&
(!namedCurrentGap || !distinguishedPrimaryIssue || JSON.stringify(premiseActors.filter(actor => actor !== primary![1]!.toLowerCase())) === JSON.stringify(headerControls)),
controlCount: premiseCount,
namedPeers: distinguishedPrimaryIssue ? headerControls : namedCurrentGap ? premiseActors.filter(actor => actor !== primary![1]!.toLowerCase()) : undefined,
remedy: {
peers: controlNames(findingStyle![1]!),
role: clauseStyle ? clauseRole : !!(roleStyle || labelledRole || (declaredStyle && designAction)),
tokens: !!style,
authority: clauseStyle ? !!clauseAuthority : !!(distinguishedStyle || (declaredStyle && designAction)),
current: [roleLabel, body].every(text => consistentRemedyEvidence(text, primary![1]!, controlNames(findingStyle![1]!))),
},
alternative: {
unresolved: !!(declaredDeferral || retainedEmphasis || retainedHierarchyGap || retainedRoleGap || unresolvedDebt || retainedPrimaryGap || retainedNoPrimary || retainedViolation),
retainedCounts: [retainedLabel?.[1], retainedControls?.[1]].filter((count): count is string => !!count).map(numberValue),
current: !cancelledRetention && !cancelledDebt && currentOptionEvidence(declined) && currentOptionEvidence(declinedLabel) &&
!/(?:^|[.!?;]\s+|\n)(?:Correction:\s*)?(?:do not|don't|never|skip|cancel|withdraw) (?:keep|leave) identical (?:header )?buttons\b/i.test(declined) &&
!closedGap.test(declined) && !invalidContract.test(declined) && !cancelledStyle.test(declined) && !withdrawnStyles.test(declined),
},
});
const retainedButtons = /^(?:❌\s*)?Keep all (two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) (?:header )?buttons identical; gap stays documented\./i.exec(remaining);
const cancelledRetainedButtons = /(?:^|[.!?;]\s+|\n)(?:Correction:\s*)?(?:do not|don't|never|skip|cancel|withdraw) (?:keep|leave) (?:all )?(two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) (?:header )?buttons identical\b/i.exec(declined);
if (headerActionIssue) {
const keep = /^[1-9]\d*[A-Z]: Keep all (two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) identical(?: \(recommended\))?$/i.exec(defer.label);
return defer !== amendment && keep && numberValue(keep[1]!) === headerControls.length + 1 &&
!conditionalHeader(declined) && !sourceAssessment.test(declined) && !withdrawn.test(declined) && !invalidContract.test(declined) &&
!closedGap.test(declined) && !cancelledRetainedButtons &&
/^(?:❌\s*)?Ships a (?:known|documented) DESIGN\.md violation; the review score stays capped and users keep scanning a flat row\./i.test(remaining);
}
return defer !== amendment && !sourceAssessment.test(declined) && !withdrawn.test(declined) && !closedGap.test(declined) && !cancelledHeaderDeferral &&
((namedTokenValid && /^(?:Leaves|Keeps|Retains) (?:a documented|the(?: documented)?) DESIGN\.md violation (?:in place|unresolved|open)\b/i.test(remaining)) ||
(retainedButtons && numberValue(retainedButtons[1]!) === otherControls + 1 &&
(!cancelledRetainedButtons || numberValue(cancelledRetainedButtons[1]!) !== otherControls + 1)) ||
/^(?:❌\s*)?(?:Leaves a documented DESIGN\.md violation in place|Keeps the documented DESIGN\.md violation and the scan problem|Violates DESIGN\.md and leaves the mis-click on [A-Za-z][A-Za-z /_-]{0,79} unaddressed|Ships a (?:known|documented) DESIGN\.md violation and the primary action (?:stays|remains) undiscoverable|Ships a header with no primary action; PLAN\.md['’]s own gap stays open|Ships the documented violation;[^.\n]*\bthe gap remains open|Primary-action ambiguity ships; documented DESIGN\.md violation remains|Decline the fix; gap stays documented and lowers the score|Decline the fix; document the violation as accepted|Keep all (?:two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) identical; record as an open DESIGN\.md violation)\b/i.test(remaining) ||
(variantRepair && /^(?:Violates DESIGN\.md and leaves users guessing which action is primary; Pass [1-7] stays at [0-9](?:\.[0-9]+)?\/10|Documented DESIGN\.md violation ships and Pass [1-7] stays at [0-9](?:\.[0-9]+)?\/10)\.$/i.test(remaining)));
});
});
return opposed.length > 0 && !!(repair || primaryRepair);
}
/** A design-system choice can name the gap without using an imperative repair verb. */
function designSystemChoiceIssue(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call || call.answered !== true || call.failed !== false || !call.sessionId || !call.toolUseId ||
call.questions.length !== 1 || !Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length ||
fp.signature !== `${call.sessionId}:${call.toolUseId}` ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0) ||
!call.answeredAt || !Number.isFinite(Date.parse(call.answeredAt))) return false;
const q = call.questions[0]!;
const lines = q.question.trim().split('\n');
const namedGap = /^(?:D[1-9]\d*\s*[—–:-]\s*)?Issue ([1-9]\d*) \(G([1-9]\d*)\): ([^?]+)\?$/.exec(lines[0]!);
const findingIssue = /^([1-9]\d*)\s*[—–:-]\s*Finding ([1-9]\d*) \(([A-Za-z][A-Za-z &/-]*)\): ([^?]+)\?$/i.exec(lines[0]!);
const fieldIssue = /^(?:D[1-9]\d*\s*[—–:-]\s*)?Issue ([1-9]\d*)(?: \(Pass ([1-7])(?:, [A-Za-z][A-Za-z &/-]*)?\))?: ([^?]+)\?$/.exec(lines[0]!) ??
(findingIssue ? [findingIssue[0], findingIssue[1], undefined, `${findingIssue[3]}: ${findingIssue[4]}`] : null);
const source = /^Project\/branch\/task: (.+)$/m.exec(q.question)?.[1] ?? '';
const sourceGaps = [...source.matchAll(/\bgap G([1-9]\d*)\b/gi)];
const ownGap = sourceGaps[0]?.[1];
const nativeIssue = fieldIssue && (q.header.trim() === `Issue ${fieldIssue[1]}` || findingIssue);
// The native Issue/option IDs own the current decision. A pass can be in
// its title or source field, and a G label is optional. If a G is present,
// another source row cannot lend this question its identity or evidence.
const scopedIssue = fieldIssue && (fieldIssue[2] || /\bPass [1-7]\b/.test(source) || nativeIssue) && sourceGaps.length <= 1 &&
[...q.question.matchAll(/\bG([1-9]\d*)\b/g)].every(m => m[1] === ownGap)
? [fieldIssue[0], fieldIssue[1], ownGap, fieldIssue[3]] : null;
const gapIssue = namedGap ?? scopedIssue;
if (gapIssue && (() => {
const [, issueNumber, gapNumber, subject] = gapIssue;
const headerIssue = /\bIssue ([1-9]\d*)\b/i.exec(q.header);
if (!q.header.trim() || (headerIssue && headerIssue[1] !== issueNumber) ||
/^(?:focus|scope|setup|routing|learnings|outside(?: design)? voices|next steps?)\b/i.test(q.header.trim()) ||
/<gstack-qid:/i.test(q.question) || q.multiSelect || q.options.length < 2 || q.options.length > 4 ||
fp.options.length !== q.options.length || !fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label) ||
!q.options.some(o => o.label === call.answers?.[q.question])) return false;
// The issue and option IDs bind a decision; its descriptive menu header
// and the wording/line count of each decision field do not supply evidence.
const ids = q.options.map(o => new RegExp(`^(${issueNumber}[A-Z])(?:[).:]?\\s+)`).exec(o.label)?.[1]);
if (ids.some(id => !id) || new Set(ids).size !== ids.length) return false;
// Count the acknowledged design decision, not optional summary formatting.
const fields = ['Project/branch/task:', 'ELI10:', 'Stakes if we pick wrong:', 'Recommendation:', 'Completeness:'];
if (q.question.includes('Net:')) fields.push('Net:');
const positions = fields.map(field => q.question.indexOf(field));
if (positions.some((position, i) => position < 0 || q.question.lastIndexOf(fields[i]!) !== position ||
(i > 0 && position <= positions[i - 1]!)) || q.question.slice(0, positions[0]).trim() !== lines[0]) return false;
const values = fields.map((field, i) => q.question.slice(positions[i]! + field.length, positions[i + 1] ?? q.question.length).trim());
const sourceOnly = /^(?:[>"“`]|Historical|Previously|Hypothetical|Quoted|Source|Archived|Earlier|Example|If|When|Once|Unless|Assuming|Provided|Pending approval)\b|^[>"“`]/i;
const inactive = /(?:^|[.!?;]\s+|\n)(?:Correction:\s*)?(?:(?:this|the) (?:finding|gap|issue|amendment|fix|decision)|G[1-9]\d*|Issue [1-9]\d*) (?:is|was|has been) (?:already |now )?["'‘“`]*(?:withdrawn|resolved|closed|superseded|hypothetical|not current|no longer current)\b|\bno current (?:defect|gap|finding|issue)\b/i;
const namedOwner = findingIssue ? `Finding ${findingIssue[2]}` : /^D[1-9]\d*/.exec(lines[0]!)?.[0];
const namedStatusPrefix = new RegExp(`(?:^|[.!?;]\\s+|\\n)(?:Correction:\\s*)?${namedOwner ?? '(?!)'} (?:is|was|has been) (?:already |now )?$`, 'i');
const namedInactive = new RegExp(namedStatusPrefix.source.replace(/\$$/, '') +
'["\'‘“`]*(?:withdrawn|resolved|closed|superseded|hypothetical|not current|no longer current)\\b', 'i');
const inactiveCurrent = (text: string) => inactive.test(text) || namedInactive.test(text);
const current = (value: string) => value.replace(/```[\s\S]*?(?:```|$)|~~~[\s\S]*?(?:~~~|$)/g, '')
.replace(/^\s*>.*$/gm, '').replace(/"[^"\n]+"|“[^”\n]+”|`[^`\n]+`|'[^'\n]+'|‘[^’\n]+’/g,
(quoted, index, source) => /^(?:withdrawn|resolved|closed|superseded|hypothetical|not current|no longer current)$/i.test(quoted.slice(1, -1)) &&
(/\b(?:(?:this|the) (?:finding|gap|issue|amendment|fix|decision)|G[1-9]\d*|Issue [1-9]\d*) (?:is|was|has been) (?:already |now )?$/i.test(source.slice(0, index)) || namedStatusPrefix.test(source.slice(0, index)))
? quoted.slice(1, -1) : '');
const sourceText = current(values[0]!.replace(/`PLAN\.md`/g, 'PLAN.md'));
// A review can name its current plan instead of repeating PLAN.md. Treat
// the title as a title only inside the review's own provenance field; a
// pass number corroborates that ownership but cannot supply it by itself.
const namedPlan = /(?:^|[,;]\s*)(?:\/plan-design-review of|reviewing|design review of)\s+(?:"([^"\n]+)"|“([^”\n]+)”|`([^`\n]+)`)(?=[,;.\s]|$)/i.exec(values[0]!);
const title = namedPlan?.slice(1).find(Boolean);
const namedCurrentPlan = !!title && !/\b[\w.-]+\.md\b/i.test(title) &&
!/\b(?:other|another|different|unrelated|foreign|historical|archived|quoted|copied|example)\b/i.test(title) &&
/\bPass [1-7]\s*\([A-Za-z][A-Za-z &/-]*\)/.test(sourceText);
const ownedSource = /\bPLAN\.md\b/.test(sourceText) ||
(!/\b[\w.-]+\.md\b/i.test(values[0]!) && (namedCurrentPlan ||
/\bPass [1-7]\s*\([A-Za-z][A-Za-z &/-]*\) of the [A-Za-z][A-Za-z -]* plan\.$/.test(sourceText)));
if (values.some(value => !value || sourceOnly.test(value)) || inactiveCurrent(current(q.question)) ||
!ownedSource || /\b(?:other|another|different|unrelated|foreign|historical|archived|quoted|copied) (?:[A-Za-z-]+ )?(?:plan|review|source)\b/i.test(sourceText) ||
/\b(?:planning|review|workflow) setup\b|\b(?:setup|onboarding|routing|posture|learnings) (?:stage|phase|step|decision)\b/i.test(current(values[0]!)) ||
(scopedIssue && !ownedSource) ||
!ids.some(id => values[3]!.startsWith(`${id} `))) return false;
const assessment = current(values[1]!);
if (/\b(?:historical|archived|hypothetical|quoted)\b|\b(?:not|isn't) (?:the )?current\b/i.test(assessment) ||
(!nativeIssue && !/\b(?:now|today|currently|proposed)\b/i.test(assessment))) return false;
// The complete comparison may live in the current native brief while
// the rendered menu uses short captions. Keep each detail block bound
// to its own native option ID; never pool evidence across alternatives.
const detailedOptions = new Map<string, string>();
const details = values[4]!.split(/\n(?:Pros\s*\/\s*cons|Options):\s*\n/i);
if (nativeIssue && details.length === 2) {
const body = details[1]!;
const starts = [...body.matchAll(/^([1-9]\d*[A-Z])[).:]\s+\S/gm)];
if (starts.length === ids.length && starts.every(start => ids.includes(start[1])) &&
new Set(starts.map(start => start[1])).size === ids.length &&
!body.split('\n').some(line => sourceOnly.test(line.trim()))) {
for (const [index, start] of starts.entries()) {
detailedOptions.set(start[1]!, body.slice(start.index, starts[index + 1]?.index ?? body.length));
}
}
}
// Recognize the fixture's design-defect classes, not a G-number or a
// prescribed sentence: ambiguous hierarchy, absent pending feedback,
// inconsistent type/spacing, or unreadable error contrast. Every class
// still needs a concrete native remedy and its own opposed open gap.
const classes: Array<{ subject: RegExp; defect: RegExp; remedy: RegExp }> = [
{ subject: /\b(?:distinguished|primary|header|hierarchy)\b/i,
defect: /\b(?:look (?:the )?(?:same|identical)|share (?:the )?same visual weight|visually identical)\b/i,
remedy: /\bfilled\b[^;\n]*#[0-9a-f]{6}[^;\n]*(?:white|black)\b[^\n]*\bghost\b/i },
{ subject: /\b(?:pending|request|loading)\b/i,
defect: /\b(?:page|request|button)\b[^.!?]*(?:just sits|freezes|no (?:visible )?(?:feedback|signal|indicator))|\b(?:shows?|gives?) no (?:pending |visible )?(?:feedback|signal|indicator)\b|\bnothing changes\b/i,
remedy: /\b(?:inline )?spinner\b[^\n]*\baria-busy\s*=\s*true\b[^\n]*\breduced.motion\b/i },
{ subject: /\b(?:type|typography|labels|headings)\b/i,
defect: /\b(?:form|labels?|type)\b[^.!?]*(?:no (?:consistent )?(?:rule|role)|inconsisten\w*|accidental|(?:three|[3-9]\d*) sizes)/i,
remedy: /\b\d+px\b[^\n]*\blabels?\b[^\n]*\b\d+px\b[^\n]*\b(?:headings?|h[1-6])\b/i },
{ subject: /\b(?:spacing|rhythm|gaps)\b/i,
defect: /\b(?:form|gaps?|spacing)\b[^.!?]*(?:no rule|without a spacing rule|random|inconsisten\w*)|\b(?:uneven|mixed|inconsistent|random) (?:spacing|gaps)\b/i,
remedy: /\bsections?\s+\d+px\b[^\n]*\bfield groups?\s+\d+px\b[^\n]*\blabel(?:\W*to\W*|\W+)(?:input|control)\s+\d+px\b/i },
{ subject: /\b(?:errors?|contrast|colou?rs?)\b/i,
defect: /\b(?:error|text|contrast)\b[^.!?]*(?:fails? WCAG|below (?:WCAG|AA)|cannot read|can't read)/i,
remedy: /#[0-9a-f]{6}\b[^\n]*#[0-9a-f]{6}\b[^\n]*\b(?:icon|text)\b/i },
];
// A numbered current Issue may state its gap in the title, then explain
// its impact in ELI10. Source/status/field ownership still apply to both.
const assertedGap = nativeIssue ? `${current(subject!)}\n${assessment}` : assessment;
const lowContrast = nativeIssue && /\b(?:error|contrast|message|text)\b/i.test(subject!) &&
[...assertedGap.matchAll(/(?:\bat\b|\babout\b|\bapproximately\b|~)\s*([0-9]+(?:\.[0-9]+)?)\s*:\s*1\b/gi)]
.some(match => Number(match[1]) < 4.5) && /\b(?:WCAG|AA)\b/.test(assessment);
const kind = classes.find((kind, index) => kind.subject.test(subject!) &&
(kind.defect.test(assertedGap) || index === 4 && lowContrast));
if (!kind || /\b(?:do not|don't|does not|doesn't) look identical\b/i.test(assessment)) return false;
return q.options.some((option, index) => {
const body = option.description?.trim() ?? '';
const detail = detailedOptions.get(ids[index]!) ?? '';
const remedy = nativeIssue ? current(`${option.label}\n${body}\n${detail}`) : current(body);
if (sourceOnly.test(body) || inactiveCurrent(remedy) || !kind.remedy.test(remedy) ||
!values[3]!.startsWith(`${ids[index]} `)) return false;
return q.options.some((other, otherIndex) => {
const declined = `${other.description?.trim() ?? ''}\n${detailedOptions.get(ids[otherIndex]!) ?? ''}`.trim();
const opposed = current(declined);
const ownedOpposition = !/\b(?:other|another|different|unrelated|foreign) (?:gap|issue|finding|decision)\b/i.test(opposed) &&
[...opposed.matchAll(/\bIssue ([1-9]\d*)\b/gi)].every(match => match[1] === issueNumber);
// A retained violation must be an affirmative current consequence,
// not words inside a prohibition or a consequence awaiting approval.
// Read the whole option so a later correction can withdraw the claim.
const retainedViolation = /(?:^|[.!?;]\s+|\n|[✅❌]\s*)(?:Leaves|Keeps) (?:the |this )?(?:plan|design|page|header) violating DESIGN\.md\b/i.test(opposed) &&
!/\b(?:not|never|no longer|cannot|can't|don't|doesn't|didn't|won't)\b[^.!?;\n]*\b(?:leaves?|keeps?) (?:the |this )?(?:plan|design|page|header) violating DESIGN\.md\b/i.test(opposed) &&
!/\b(?:if|when|once|unless|assuming|provided|pending|contingent|conditional)\b|\b(?:before|after|requires?|needs?|subject to|depends? on)\s+(?:(?:user|later|further|your|owner|explicit)\s+)?(?:approval|acceptance)\b/i.test(opposed);
return ownedOpposition && other !== option && /^(?:Keep|Leave|Defer|Decline|No)\b/i.test(other.label.replace(new RegExp(`^${ids[otherIndex]}[).:]?\\s+`), '')) &&
!sourceOnly.test(declined) && !inactiveCurrent(current(declined)) &&
!/\b(?:(?:does?|did) not|no longer|never) violates? DESIGN\.md\b/i.test(opposed) &&
(new RegExp(`\\b(?:gap\\s+)?G${gapNumber}\\s+(?:stays|remains|is)\\s+(?:open|unresolved)\\b`, 'i').test(current(declined)) ||
(!!scopedIssue && (/\b(?:the |[a-z-]+ )?gap (?:stays|remains|is) (?:open|unresolved)\b/i.test(current(declined)) ||
(!!nativeIssue && /\b(?:known|documented) (?:WCAG )?AA failure ships\b/i.test(current(declined)) && /\bstays open\b/i.test(current(declined))) ||
(!!nativeIssue && /\b(?:stays|remains) (?:open|unresolved)\b/i.test(current(declined)) &&
!/\b(?:other|another|different|unrelated) (?:gap|issue|finding|decision)\b/i.test(current(declined)) &&
[...current(declined).matchAll(/\bIssue ([1-9]\d*)\b/gi)].every(match => match[1] === issueNumber)) ||
(!!nativeIssue && /^Record as unresolved[.;]/i.test(current(declined))) ||
// A kept violation may name the relevant contract, rather than
// use the exact phrase "gap remains open". It still belongs to
// this issue and the same design-defect class as the remedy.
(!!nativeIssue && /\bviolates DESIGN\.md(?:'s)?\b/i.test(opposed) && kind.subject.test(opposed) &&
!/\b(?:(?:does?|did) not|no longer|never) violates?\b|\b(?:historical|previous|earlier|example|quoted|hypothetical)\b/i.test(opposed)) ||
(!!nativeIssue && /\bviolates DESIGN\.md(?:'s)? (?:stated |existing |documented )?(?:primary treatment|two-role rule|spacing scale|contrast requirement)\b/i.test(current(declined))) ||
/\b(?:plan|design|page|header)\b[^.!?]*\b(?:keeps|retains|leaves|ships)\b[^.!?]*\bDESIGN\.md violation\b/i.test(current(declined)) ||
retainedViolation) &&
[...q.options.flatMap(o => [...`${o.label} ${o.description ?? ''}`.matchAll(/\bG([1-9]\d*)\b/g)])]
.every(m => m[1] === gapNumber)));
});
});
})()) return true;
const issue = /^(?:D[1-9]\d*\s*[—–:-]\s*)?Issue ([1-9]\d*): (.+)\?$/.exec(lines[0]!);
if (!issue || q.header.trim() !== `Issue ${issue[1]}` || lines.length !== 7 ||
!/^Project\/branch\/task: [^\n,]+ on [^\n,]+, PLAN\.md design review, Pass [1-7] [A-Za-z][A-Za-z &()-]+\.$/.test(lines[1]!) ||
!/^ELI10: \S/.test(lines[2]!) || !/\bDESIGN\.md\b/.test(lines[2]!) ||
!/^Stakes if we pick wrong: \S/.test(lines[3]!) || !/^Recommendation: \S/.test(lines[4]!) ||
!/^Completeness: \S/.test(lines[5]!) || !/^Net: \S/.test(lines[6]!) ||
/<gstack-qid:|```|^ELI10: (?:Example|Hypothetical|Quoted)\b/im.test(q.question) || q.multiSelect ||
q.options.length < 2 || q.options.length > 4 || new Set(q.options.map(o => o.label)).size !== q.options.length ||
!q.options.every(o => new RegExp(`^${issue[1]}[A-Z]: \\S`).test(o.label)) ||
fp.options.length !== q.options.length || !fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label) ||
!q.options.some(o => o.label === call.answers?.[q.question])) return false;
// These are current visual/interaction choices, not reviewer participation or next-step routing.
const subjects = [
/^How should [A-Z][A-Za-z0-9 _/-]{0,79} be distinguished from [A-Z][A-Za-z0-9 ,/_-]{0,119}$/,
/^What does the user see while [A-Z][A-Za-z0-9 _/-]{0,79} is pending for [1-9]\d*(?:[-–][1-9]\d*)? seconds$/,
/^What type scale should (?:form )?labels(?: and section headings)? use$/,
/^What vertical spacing rhythm should the form use$/,
/^How should (?:the )?error message meet WCAG AA contrast$/,
];
const subject = subjects.findIndex(pattern => pattern.test(issue[2]!));
if (subject < 0) return false;
const assessments = [/^ELI10: The header shows\b/, /^ELI10: After clicking\b/,
/^ELI10: Labels on the form are set\b/, /^ELI10: Gaps between sections are\b/, /^ELI10: The error message is\b/];
if (!assessments[subject]!.test(lines[2]!) ||
/(?:^|[.!?]\s+)(?:This (?:issue|finding) (?:is|has been) (?:withdrawn|resolved|closed)|We have (?:resolved|closed|withdrawn) this (?:issue|finding)|No current (?:issue|finding|gap|defect|violation) (?:remains|exists))\b/i.test(lines[2]!.slice(7))) return false;
const control = /^How should (.+) be distinguished from /.exec(issue[2]!)?.[1];
const concrete = [new RegExp(`^${control}\\b[^\\n]*\\b(?:filled|ghost|outlined|primary)\\b`, 'i'),
/^(?:Spinner|InlineStatus|Static indicator)\b/i, /^[1-9]\d*px\b/i, /^[1-9]\d*px\b/i, /^#[0-9a-f]{6}\b/i][subject]!;
const conforming = q.options.filter(o => concrete.test(o.label.replace(/^[1-9]\d*[A-Z]: /, '')) &&
(/^✅ Exact(?:ly)? (?:the (?:two )?)?DESIGN\.md\b/.test(o.description ?? '') ||
(subject === 0 && new RegExp(`^✅ ${control} is [^\\n]+\\bexactly per DESIGN\\.md\\b`).test(o.description ?? ''))));
return conforming.some(choice => q.options.some(o => o !== choice &&
/^❌ (?:Ships (?:the documented violation|a known WCAG AA failure)\b|Deviates from the DESIGN\.md\b)/m.test(o.description ?? '')));
}
/** Named decision fields may be compact prose; native choices still own the finding. */
function compactPrimaryDecision(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call || call.answered !== true || call.failed !== false || !call.sessionId || !call.toolUseId ||
!call.answeredAt || !Number.isFinite(Date.parse(call.answeredAt)) || call.questions.length !== 1 ||
!Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length ||
fp.signature !== `${call.sessionId}:${call.toolUseId}` ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0)) return false;
const q = call.questions[0]!;
if (q.multiSelect || q.options.length !== 2 || new Set(q.options.map(o => o.label)).size !== 2 ||
fp.options.length !== 2 || !fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label) ||
!q.options.some(o => o.label === call.answers?.[q.question]) || /<gstack-qid:/i.test(q.question)) return false;
const headline = /^(?:D[1-9]\d*\s*[—–:-]\s*)?Issue ([1-9]\d*): ([A-Za-z][A-Za-z0-9 _-]{0,39}) (?:has no|lacks) primary[- ]action hierarchy\./i.exec(q.question.trim());
if (!headline || q.header.trim() !== `Issue ${headline[1]}`) return false;
const issue = headline[1]!, control = headline[2]!;
const inactive = 'withdrawn|superseded|resolved|closed|hypothetical|unproven|rejected|cancelled|canceled|deferred|not current|no longer current';
const owner = `(?:This (?:issue|finding|question|amendment|deferral|style|fix|remedy|choice|option)|Issue ${issue}|(?:These|The|This) (?:tokens?|styles?|primary treatment))`;
const boundary = '(?:^|[.!?;]\\s+|\\n)(?:Correction:\\s*)?';
const scalarPrefix = new RegExp(`${boundary}${owner} (?:is|was|are|were|has been|have been) $`, 'i');
const current = (value: string) => value
.replace(/```[\s\S]*?(?:```|$)|~~~[\s\S]*?(?:~~~|$)/g, '')
.replace(/^(?:\s*>| {4}|\t).*$/gm, '')
.replace(/`[^`\n]*`|"[^"\n]*"|“[^”\n]*”|(?<!\w)'[^'\n]*'(?!\w)|‘[^’\n]*’/g, (quoted, index, source) =>
new RegExp(`^(?:${inactive})$`, 'i').test(quoted.slice(1, -1)) && scalarPrefix.test(source.slice(0, index)) ? quoted.slice(1, -1) : '')
.replace(/\*\*/g, '');
const invalid = (value: string) =>
new RegExp(`${boundary}${owner} (?:is|was|are|were|has been|have been) (?:${inactive})\\b`, 'i').test(value) ||
new RegExp(`${boundary}(?:(?:This|The) (?:gap|violation) (?:is|was|has been) (?:already |now )?(?:resolved|fixed|closed)|No current (?:gap|issue|finding|violation) (?:remains|exists))\\b`, 'i').test(value) ||
new RegExp(`${boundary}(?:If|When|Once|Provided|Assuming|Pending) (?:approval|approved|acceptance|accepted|(?:we|you) (?:approve|accept))\\b`, 'i').test(value) ||
new RegExp(`${boundary}(?:Do not|Don't|Never|Skip|Cancel|Withdraw) (?:apply|use|add|keep) (?:this (?:fix|amendment)|(?:the |these )?(?:tokens?|styles?|primary treatment))\\b`, 'i').test(value) ||
new RegExp(`${boundary}(?:${control} (?:already (?:is|has)|is already) (?:the (?:only |visible )?primary action|primary[- ]action hierarchy)|This (?:issue|finding) has no current (?:gap|defect)|(?:This|The) (?:amendment|fix) keeps (?:all )?(?:[a-z]+|[1-9]\\d*) buttons identical)\\b`, 'i').test(value) ||
/(?:^|[.!?;]\s+|\n)(?:Historical|Hypothetical|Quoted|Source|Archived|Example)(?:\s+(?:review|example|excerpt|assessment|material|text))?\s*:/i.test(value);
const text = current(q.question);
if (invalid(text)) return false;
// These are the skill's existing decision fields, not a particular sentence
// or line layout. Duplicate/missing fields cannot borrow a neighboring issue.
const fields = ['Project/branch/task:', 'ELI10:', 'Stakes if we pick wrong:', 'Recommendation:', 'Completeness:', 'Net:'];
const positions = fields.map(field => text.indexOf(field));
if (positions.some((position, i) => position < 0 || text.lastIndexOf(fields[i]!) !== position ||
(i > 0 && position <= positions[i - 1]!)) ||
text.slice(0, positions[0]).trim() !== headline[0] ||
(text.match(/\?/g)?.length ?? 0) !== 1 || !/\?\s*$/.test(text)) return false;
const values = fields.map((field, i) => text.slice(positions[i]! + field.length, positions[i + 1] ?? text.length).trim());
if (values.some(value => !value) || values.some(value => /^(?:If|When|Once|Unless|Assuming|Provided|Historical|Hypothetical|Quoted|Source|Example)\b/i.test(value)) ||
!/\bDESIGN\.md\b/.test(values[3]!)) return false;
const count = (value: string) => /^\d+$/.test(value) ? Number(value) :
['zero', 'one', 'two', 'three', 'four', 'five', 'six', 'seven', 'eight', 'nine', 'ten'].indexOf(value.toLowerCase());
const names = (value: string) => value.toLowerCase().split(/\s*[,/]\s*(?:and\s+)?|\s+and\s+/).map(s => s.trim()).sort();
const same = (a: string[], b: string[]) => JSON.stringify(a) === JSON.stringify(b);
const assessment = /^The header shows ([A-Za-z][A-Za-z0-9 ,/_-]{0,159}) as (two|three|four|five|six|seven|eight|nine|ten|[1-9]\d*) identical buttons\./i.exec(values[1]!);
if (!assessment) return false;
const actors = names(assessment[1]!);
if (new Set(actors).size !== actors.length || actors.length !== count(assessment[2]!) || !actors.includes(control.toLowerCase())) return false;
const peers = actors.filter(actor => actor !== control.toLowerCase());
const ids = q.options.map(o => new RegExp(`^(${issue}[A-Z])[).:]\\s+`).exec(o.label)?.[1]);
if (ids.some(id => !id) || new Set(ids).size !== 2 || !ids.some(id => values[3]!.startsWith(`${id} `))) return false;
const offered = ids.map(id => [...values[4]!.matchAll(new RegExp(`(?:^|\\s)${id}[).:]\\s+`, 'g'))]);
if (offered.some(matches => matches.length !== 1)) return false;
return q.options.some((option, index) => {
const body = current(option.description ?? ''), other = q.options[1 - index]!, declined = current(other.description ?? '');
const style = /^([A-Za-z][A-Za-z0-9 _-]{0,39}): filled (#[0-9a-f]{6}) with (white|black) text\. ([A-Za-z][A-Za-z0-9 ,/_-]{0,159}): neutral ghost(?: buttons)?\./i.exec(body);
const keep = new RegExp(`^${ids[1 - index]}[).:] Keep (two|three|four|five|six|seven|eight|nine|ten|[1-9]\\d*) equal buttons(?: \\(recommended\\))?$`, 'i').exec(other.label);
if (!style || style[1]!.toLowerCase() !== control.toLowerCase() || !same(names(style[4]!), peers) ||
!new RegExp(`^${ids[index]}[).:] Filled primary ${control}(?: \\(recommended\\))?$`, 'i').test(option.label) ||
!keep || count(keep[1]!) !== actors.length || invalid(body) || invalid(declined) ||
!/^No change\. Documented as a declined fix; Pass [1-7] stays below 10\./i.test(declined)) return false;
// The detailed offered action must agree with its native menu's tokens and
// actors; prose about another control cannot lend this choice a remedy.
const start = offered[index]![0]!.index!, next = offered[1 - index]![0]!.index!;
const action = values[4]!.slice(start, next > start ? next : undefined);
const detail = new RegExp(`(?:^|[✅]\\s*)${control} becomes the only filled button \\((#[0-9a-f]{6}), (white|black) text\\); ([A-Za-z][A-Za-z0-9 ,/_-]{0,159}) become neutral ghost buttons`, 'i').exec(action);
return !!detail && detail[1]!.toLowerCase() === style[2]!.toLowerCase() && detail[2]!.toLowerCase() === style[3]!.toLowerCase() && same(names(detail[3]!), peers);
});
}
/** A completed finding can start the passes when the caller already supplied the focus. */
export function isDesignCountFirstReview(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.answered || call.failed) return false;
if (isDesignCountSetup(fp)) return false;
if (numberedVisualHierarchyFinding(fp)) return true;
const findingScope = { primary: false, ownsPrimaryPremise: false };
if (ordinaryDesignIssue(fp, findingScope)) return true;
if (findingScope.ownsPrimaryPremise) return false;
// A complete native decision can supply its own source, current defect,
// remedy and opposition in review fields. Validate those independently
// before closing the loose marker fallback for recognized primary issues.
if (designSystemChoiceIssue(fp)) return true;
if (findingScope.primary) return false;
if (compactPrimaryDecision(fp)) return true;
if (designFirstReviewAUQ(fp)) return true;
return call.questions.some(q => {
if (!call.answers?.[q.question] || q.options.length < 2) return false;
if (DESIGN_SETUP_HEADER.test(q.header.trim())) return false;
const id = questionId(q.question);
if (DESIGN_SETUP_ID.test(id)) return false;
// Native fingerprints prepend the menu header. Inspect the actual question
// for an explicit finding that offers a plan amendment and deferral.
if (call.answered === true && call.failed === false && /^Pass\s*[1-7]\s*\([^)]*\)\s*[—–:]\s*Finding\s*[1-9]\d*:\s+\S/i.test(q.question.trim()) &&
/^plan-design-review-[a-z0-9-]+$/i.test(id) &&
(q.question.match(/<gstack-qid/gi)?.length ?? 0) === 1 &&
/\b(?:Apply|Add|Fix|Specify|Define|Restore)\b[^?\n]*\b(?:to|in) the plan\?\s*<gstack-qid:[^>]+>\s*$/i.test(q.question) &&
q.options.some(option => /^(?:Apply|Add|Fix|Specify|Define|Restore)\b/i.test(option.label)) &&
q.options.some(option => /^(?:Defer|Leave|Keep as-is|Accept the gap)\b/i.test(option.label)) &&
q.options.some(option => option.label === call.answers?.[q.question]) &&
Array.isArray(call.unansweredQuestionIndices) && !call.unansweredQuestionIndices.length &&
fp.signature === `${call.sessionId}:${call.toolUseId}`) return true;
// A named or scored pass can ask for a missing design requirement before a
// numbered finding heading appears. Its actual decision and opposed
// choices establish review; a score or familiar qid alone cannot.
const scoredPass = /^(?:D\s*\d+\s*[—–:-]\s*)?Pass\s*[1-7]\s*\([^)]*\)\s*[—–:-]\s*(?:10|[0-9])(?:\.[0-9]+)?\/10[.!:]/i.test(q.question.trim());
const namedPass = /^(?:D\s*\d+\s*[—–:-]\s*)?Pass\s*[1-7]\s*[—–:-]\s*[A-Za-z][A-Za-z ]{3,60}:\s+/i.test(q.question.trim());
const chosen = q.options.some(option => option.label === call.answers?.[q.question]);
const fixChoice = q.options.some(option => /^(?:Add|Fix|Specify|Define|Restore)\b/i.test(option.label));
const leaveChoice = q.options.some(option => /^(?:Leave as-is|Keep as-is|Defer|Accept the gap)\b/i.test(option.label) ||
/^Skip\s*[—–-]\s*implied by\s+[^.!?]+\bgap$/i.test(option.label));
if ((scoredPass || namedPass) && /^plan-design-review-[a-z0-9-]+$/i.test(id) &&
(q.question.match(/<gstack-qid/gi)?.length ?? 0) === 1 &&
/\b(?:gap|problem|defect|missing|inconsisten\w*)\b|\b(?:doesn['’]t|does not)\s+(?:record|specify|define|describe)\b/i.test(q.question) &&
/\bShould I (?:add|fix|specify|define|restore)\b[^?]+\?\s*<gstack-qid:[^>]+>\s*$/i.test(q.question) &&
fixChoice && leaveChoice && chosen &&
!(call.unansweredQuestionIndices?.length) &&
fp.signature === `${call.sessionId}:${call.toolUseId}`) return true;
// Native pass decisions can carry a D-number before the pass title and
// use plan-design-passN rather than plan-design-review-... identities.
// Bind both forms to the same explicit pass and an offered choice that
// leaves a named gap unresolved. Pass readiness is only setup.
const numberedPass = /^D\s*\d+\s*[—–:-]\s*Pass\s*([1-7])\s*\([^)]*\)\s*:/i.exec(q.question.trim());
const passId = /^plan-design-pass([1-7])-/i.exec(id);
const unresolvedChoice = q.options.some(option =>
/\b(?:leave|keep|defer|accept)\b/i.test(option.label) &&
/\b(?:gap|problem|defect|inconsisten\w*)\b/i.test(`${option.label} ${option.description ?? ''}`));
if (numberedPass && passId && numberedPass[1] === passId[1] && unresolvedChoice && /\?/.test(q.question)) return true;
// These are issue-bearing pass statements in actual answered calls,
// not a setup request that merely mentions the seven review passes.
return /^Pass\s*[1-7]\s+(?:surfaces|(?:also\s+)?(?:found|flagged))\b/i.test(q.question.trim()) &&
/\?/.test(q.question);
});
}
/** Setup by structure: the recognized setup packet, or a native call whose every
* question carries a setup header or setup question ID. */
export function isDesignCountStructuralSetup(fp: AskUserQuestionFingerprint): boolean {
if (isDesignCountSetup(fp)) return true;
const call = fp.nativeCall;
return !!call && call.questions.length > 0 && call.questions.every(q =>
DESIGN_SETUP_HEADER.test(q.header.trim()) || DESIGN_SETUP_ID.test(questionId(q.question)));
}
/** The review's TODO contract offers exactly A) Add to TODOS.md, B) Skip, C) Build it now. */
export function isDesignTodoProposal(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.answered || call.failed || call.questions.length !== 1) return false;
const labels = call.questions[0]!.options.map(option => option.label.trim()
.replace(/^\d*[A-C][).:]?\s+/, '').replace(/\s*\(recommended\)\s*$/i, ''));
return labels.length === 3 && /^Add to TODOS\.md\b/i.test(labels[0]!) && /^Skip\b/i.test(labels[1]!) &&
/^Build it now\b/i.test(labels[2]!);
}
/** After setup, the first answered native decision that is not setup, a TODO
* proposal, a completion handoff or artifact rendering starts review. Handoff and
* artifact calls are classified before this predicate runs. */
export function isDesignCountReviewStart(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
return !!call && call.answered === true && call.failed === false && fp.signature === `${call.sessionId}:${call.toolUseId}` &&
call.questions.length > 0 && call.questions.every(q => Boolean(call.answers?.[q.question])) &&
!isDesignCountStructuralSetup(fp) && !isDesignTodoProposal(fp);
}
/** A closed recap may explain why Eng is next; it cannot request another fix. */
function closedDesignGateRecap(tail: string, descriptions: string[]): boolean {
const navigation = /\bWhat(?:['’]s)?\s+next\?\s*<gstack-qid:[a-z0-9-]+>\s*$/i.exec(tail);
if (!navigation) return false;
const body = tail.slice(0, navigation.index).trim();
const gate = /^(?:Eng(?:ineering)? Review is (?:the )?required (?:shipping gate|gate before shipping))[.!]?$/i;
const sentences = (text: string) => text.split(/[.!]\s+|[.!]$/).map(s => s.trim()).filter(Boolean);
const recap = (text: string): boolean => {
// Each count describes completed or explicitly absent work. A positive
// deferred/open count is not a closed review, regardless of its title.
const count = /^(?:(?:\d+|all)\s+(?:design\s+)?(?:decisions|findings|issues)\s+(?:(?:are|were)\s+)?(?:resolved|approved|addressed|closed)|\d+\s+(?:implementation\s+)?tasks\s+(?:(?:are|were)\s+)?(?:added|recorded|ready)|(?:no|zero|0)\s+(?:deferred(?:\s+(?:decisions|findings|issues|tasks|items))?|(?:unresolved|open|pending|outstanding)\s+(?:decisions|findings|issues|tasks|items)))$/i;
if (text.split(/,\s*(?:and\s+)?|\s+and\s+/i).every(part => count.test(part))) return true;
// Only a declarative completed-review subject can introduce explanatory
// content. Separate clauses, questions and conditional/future work fail.
if (!/^(?:The|This)\s+(?:design\s+)?review\s+(?:has\s+)?(?:added|recorded|approved|addressed|specified|covered|resolved)\s+\S/i.test(text)) return false;
if (/[;?<>]|\b(?:if|unless|until|once|when|should|must|need|needs|will|would|could|please|then|also|still|missing|unresolved)\b|\b(?:and|but)\s+(?:first\s+)?(?:do|add|fix|repair|implement|resolve|decide|configure|remove|delete|pick|choose)\b/i.test(text)) return false;
const clauses = text.split(/\s+[—–]\s+/);
return clauses.length <= 2 && (clauses.length === 1 || /^(?:architectural|engineering|implementation)\s+(?:implications|considerations|details)\b/i.test(clauses[1]!));
};
const parts = sentences(body);
if (parts.filter(part => gate.test(part)).length !== 1 ||
!parts.every(part => gate.test(part) || recap(part))) return false;
return descriptions.every(description => sentences(description).every(part =>
gate.test(part) || recap(part) ||
/^Exit plan mode and proceed on your own$/i.test(part) ||
/^You have \d+ (?:concrete )?(?:implementation )?tasks ready to build from$/i.test(part)));
}
/** A qidless closed handoff must consume every question/description clause. */
function resolvedDesignHandoff(q: NonNullable<AskUserQuestionFingerprint['nativeCall']>['questions'][number]): number | null {
if (!/^next review$/i.test(q.header.trim()) || q.options.length !== 2) return null;
const completed = /^Design review complete [—–-] (?:10|[0-9](?:\.\d+)?)\/10 (?:→|->) (?:10|[0-9](?:\.\d+)?)\/10\. All ([1-9]\d*) decisions resolved\. The plan is design-complete; next is the required shipping gate\. What['’]s next\?$/.exec(q.question.trim());
if (!completed) return null;
const labels = q.options.map(o => o.label.trim().replace(/\s*\(recommended\)\s*$/i, ''));
const review = labels.findIndex(label => /^Run \/plan-eng-review$/i.test(label));
const manual = labels.findIndex(label => /^Skip\s*[—–-]\s*I['’]ll handle next steps manually$/i.test(label));
if (review < 0 || manual < 0 || review === manual) return null;
const description = (index: number) => (q.options[index]!.description ?? '').trim().replace(/\s+/g, ' ');
const topics = '(?:spinner|skeleton|(?:button|switch|field) (?:keyboard|focus|loading|error|disabled|pending|success)|(?:keyboard|focus|loading|error|disabled|pending|success) (?:states?|behavior|navigation))';
const recap = new RegExp('^Eng review is the required shipping gate\\. It validates architecture, component wiring, tests, and accessibility implementation against the ' + completed[1] + ' approved design decisions\\. This design review added interaction specs \\(' + topics + '(?:, ' + topics + ')*\\), so eng review needs to validate their architectural fit\\.$');
if (!recap.test(description(review)) ||
!/^End the review workflow here\. The improved plan is at the (?:e2e output|approved plan) path; implementation can begin\. Run \/plan-eng-review later before shipping\.$/.test(description(manual))) return null;
return manual + 1;
}
function designHandoff(fp: AskUserQuestionFingerprint): { manualIndex: number | null } | null {
const call = fp.nativeCall;
if (!call || call.failed || call.questions.length !== 1 ||
fp.signature !== `${call.sessionId}:${call.toolUseId}`) return null;
const q = call.questions[0]!;
if (q.multiSelect || q.options.length < 2) return null;
const pending = call.answered === false && call.answers === undefined && call.answeredAt === undefined &&
(call.unansweredQuestionIndices === undefined || (Array.isArray(call.unansweredQuestionIndices) &&
call.unansweredQuestionIndices.length === 1 && call.unansweredQuestionIndices[0] === 0));
const resolvedManual = call.failed === false && (call.answered === true || pending) ? resolvedDesignHandoff(q) : null;
if (resolvedManual !== null) return { manualIndex: resolvedManual };
if (!/^next\s+steps?$/i.test(q.header.trim())) return null;
const ids = [...q.question.matchAll(/<gstack-qid:\s*([a-z0-9-]+)\s*>/gi)].map(match => match[1]);
if ((q.question.match(/<gstack-qid\b/gi) ?? []).length !== 1 || ids.length !== 1 || !/^plan-design-(?:review-)?next-steps?$/i.test(ids[0]!)) return null;
const declaration = q.question.trim().replace(/^D\s*\d+\s*[—–:-]\s*/i, '')
.replace(/^next\s+steps?\s*:\s*/i, '');
// Scores and a completed decision count describe a closed review. A
// condition or unresolved gap cannot masquerade as its next-step menu.
const completed = /^Design\s+review\s+(?:is\s+)?complete(?:[.!]|\s+\((?:\d+(?:\.\d+)?(?:\/10)?\s*(?:→|->|to)\s*)?\d+(?:\.\d+)?\/10(?:,\s*\d+\s+decisions?(?:\s+(?:made|added))?)?\)[.!])(?:\s|$)/i.exec(declaration);
if (!completed) return null;
const requiredGateOffer = /^The required next gate is Eng(?:ineering)? Review\s*[—–-]\s*want me to run it now\?\s*<gstack-qid:[a-z0-9-]+>\s*$/i.test(declaration.slice(completed[0].length).trim());
const closedRecap = q.options.length === 2 && closedDesignGateRecap(
declaration.slice(completed[0].length).trim(), q.options.map(option => option.description ?? ''));
const requiredGateQuestion = requiredGateOffer || closedRecap || /^(?:\d+ implementation tasks ready\.\s*)?Eng(?:ineering)? Review is the required shipping gate\.\s*What next\?\s*<gstack-qid:[a-z0-9-]+>\s*$/i.test(declaration.slice(completed[0].length).trim());
// The offered Eng action can carry the required-gate declaration while the
// closed question asks only what is next. Its descriptions remain part of
// the decision, so they cannot conceal a new repair or conditional closure.
const describedRequiredGate = /^What['’]s\s+next\?\s*<gstack-qid:[a-z0-9-]+>\s*$/i.test(declaration.slice(completed[0].length).trim()) &&
q.options.some(option => /^Run \/plan-eng-review(?:\s*\(recommended\))?$/i.test(option.label.trim()) &&
/^Required gate before shipping[.!]/i.test(option.description ?? ''));
const guardedNavigation = requiredGateQuestion || describedRequiredGate;
if (!requiredGateQuestion && !/\bWhat['’]s\s+next\?\s*<gstack-qid:[a-z0-9-]+>\s*$/i.test(declaration)) return null;
// A routing label cannot conceal a new repair in its description.
if (guardedNavigation && q.options.some(option =>
/(?:^|[.!?;]\s+|\b(?:proceed to|continue to|must|need to)\s+)(?:(?:please|first|then|also)\s+)*(?:add|fix|repair|implement|resolve|decide)\b|\b(?:(?:should|could|can|would)\s+(?:we|I)|(?:we|I)\s+(?:should|could|can|would))\s+(?:add|fix|repair|implement|resolve|decide)\b/i.test(option.description ?? '') ||
/\b(?:Design|the|this)\s+review\s+(?:(?:is|remains)\s+)?(?:not\s+(?:complete|done|resolved)|incomplete|unfinished)\b|\bnot\s+all\s+(?:decisions|findings|issues|gaps)\s+(?:are\s+)?(?:resolved|complete|done)\b|\b(?:decisions|findings|issues|gaps)\s+(?:are\s+)?not\s+(?:resolved|complete|done)\b/i.test(option.description ?? '') ||
/\b(?:once|after|when|if|unless|until)\b[^.!?]*\b(?:review|decisions?|findings?|issues?|gaps?)\b[^.!?]*\b(?:complete|done|resolved)\b|\b(?:review|decisions?|findings?|issues?|gaps?)\b[^.!?]*\b(?:complete|done|resolved)\b[^.!?]*\b(?:once|after|when|if|unless|until)\b/i.test(option.description ?? ''))) return null;
// A closed heading does not override an affirmative outstanding-work claim
// in its recap. Zero/no outstanding work is a compatible completion claim.
const outstanding = (guardedNavigation ? [declaration, ...q.options.map(o => o.description ?? '')].join('\n') : declaration)
.replace(/\b(?:no|zero|0)\s+(?:unresolved|open|pending|unaddressed|remaining|outstanding)\s+(?:[a-z-]+\s+){0,3}(?:gaps?|issues?|decisions?|requirements?|work)\b/gi, '')
.replace(/\bno\s+(?:gaps?|issues?|decisions?|requirements?|work)\s+remains?\b/gi, '');
if (/\b(?:unresolved|open|pending|unaddressed|remaining|outstanding)\s+(?:[a-z-]+\s+){0,3}(?:gaps?|issues?|decisions?|requirements?|work)\b|\b(?:gaps?|issues?|decisions?|requirements?|work)\s+(?:still\s+)?remains?\b|\b(?:gaps?|issues?|decisions?|requirements?|work)\s+(?:is|are)\s+still\s+(?:unresolved|open|pending|unaddressed)\b/i.test(outstanding)) return null;
const labels = q.options.map(o => o.label.trim().replace(/^[A-Z][).]\s*/i, '')
.replace(/\s*\(recommended\)\s*$/i, '').trim());
const manual = labels.map(label => /^(?:Handle next steps manually|Skip\s*[—–-]\s*I['’]ll handle next steps manually)$/i.test(label) ||
(guardedNavigation && /^Skip\s*[—–-]\s*handle (?:next steps )?manually$/i.test(label)));
const review = labels.map(label => /^Run \/plan-eng-review(?: next)?(?: \(required gate\))?$/i.test(label));
const navigation = labels.map(label => /^(?:Skip to implementation|Run \/plan-ceo-review(?: first)?|Run \/design-(?:shotgun|html))$/i.test(label));
if (manual.filter(Boolean).length > 1 || !review.some(Boolean) ||
!labels.every((_, i) => manual[i] || review[i] || navigation[i])) return null;
// Classification does not invent a missing stop option. Only an offered
// manual action can steer a pending question away from another workflow.
const index = manual.findIndex(Boolean);
return { manualIndex: index < 0 ? null : index + 1 };
}
/** Completed handoffs retain raw evidence and their own administrative count. */
export function isDesignCompletionHandoff(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.answered || call.failed || !Array.isArray(call.unansweredQuestionIndices) ||
call.unansweredQuestionIndices.length || designHandoff(fp) === null) return false;
const q = call.questions[0]!;
return q.options.some(option => call.answers?.[q.question] === option.label);
}
/** Preserve the native-only outside opt-out, then finish this review at its actual handoff. */
export function pickDesignCountQuestion(
routing: AskUserQuestionFingerprint,
active: AskUserQuestionFingerprint,
): number | null {
const outside = pickDesignCountOutsideVoices(routing, active);
if (outside !== null) return outside;
return active.nativeCall?.answered ? null : designHandoff(active)?.manualIndex ?? null;
}
-27
View File
@@ -1,27 +0,0 @@
import type { AskUserQuestionFingerprint } from './claude-pty-runner';
import { isDesignCountFirstReview } from './design-count-review';
export function isDesignUIScopeReview(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.answered || call.failed || !Array.isArray(call.unansweredQuestionIndices) ||
call.unansweredQuestionIndices.length || !call.questions.length ||
fp.signature !== `${call.sessionId}:${call.toolUseId}`) return false;
if (call.questions.some(q => q.multiSelect || q.options.length < 2 ||
new Set(q.options.map(option => option.label)).size !== q.options.length ||
!q.options.some(option => option.label === call.answers?.[q.question]))) return false;
if (isDesignCountFirstReview(fp)) return true;
const workflow = /\b(?:review(?:s|ers?)?|scope|setup|learnings|routing|mockups?|permissions?|codex|claude|outside)\b/i;
const ui = /\b(?:dashboard|hierarchy|panels?|layout|headers?|buttons?|navigation|notifications?|activity|actions?|spacing|colou?rs?|fonts?|typography|loading|errors?|focus|contrast|keyboard|mobile|responsive|toasts?|modals?|empty)\b/i;
return call.questions.some(q => {
if (/^(?:scope|focus|learnings|routing|next steps?|outside(?: design)? voices)$/i.test(q.header.trim())) return false;
const issue = /^(?:D\d+\s*[—–:-]\s*)?Issue ([1-9]\d*)\s*[:—–-]\s*([^\n]+\?)$/i.exec(q.question.split('\n')[0]!.trim());
if (!issue || workflow.test(issue[2]!) || !ui.test(issue[2]!) ||
!q.options.some(option => ui.test(`${option.label} ${option.description ?? ''}`))) return false;
const context = /^Project\/branch\/task:([^\n]*)/mi.exec(q.question)?.[1] ?? '';
const namedPlans = context.match(/\b[\w.-]+\.md\b/gi) ?? [];
if ((namedPlans.length && !namedPlans.some(plan => /^PLAN\.md$/i.test(plan))) ||
/\b(?:before|prior to)\s+Pass\b/i.test(context)) return false;
const choice = new RegExp(`^${issue[1]}[A-Z](?:[).:—–-]\\s*|\\s+)\\S`);
return q.options.every(option => choice.test(option.label) && !workflow.test(option.label));
});
}
-682
View File
@@ -1,682 +0,0 @@
import type { AskUserQuestionFingerprint } from './claude-pty-runner';
import type { NativePlanQuestionCall } from './plan-count-transcript';
/** Concrete product decisions, separate from the skill's mandatory Step-0 confirmations. */
export const DEVEX_COUNT_FILES: Record<string, string> = {
'README.md': `# EvalKit SDK
EvalKit is a Python SDK for ML engineers evaluating LLM responses. The primary
developer writes Python daily, uses a terminal, and wants a local result before
connecting the SDK to production CI. The agreed review posture is DX POLISH:
improve the existing SDK's touchpoints within the beta release scope.
## Getting started
Install with \`python -m pip install evalkit==2.0.0b1\`,
then follow the quickstart's command: \`python examples/first_eval.py\`.
The published package inventory is in docs/package-contents.txt.
The chosen first-success experience is an included, copy-paste demo command:
\`python -m evalkit.demo\`. It evaluates bundled sample responses and prints
real per-example scores plus an overall score. It needs no hosted playground
or new interactive UI. Like every first evaluation, it currently waits for the
mandatory CI check described in docs/current-contracts.md.
The bundled demo already works without a developer API key. Its sample evaluation
uses the shipped mock transport; its mandatory remote CI check uses the included
sample-project binding. No credentials step precedes this first demo result.
The keyless demo still waits for that CI check and has no skip or offline bypass.
After the demo, developers obtain a key for their first live evaluation at
https://console.evalkit.example/settings/api-keys: select the project, choose
Create key, copy the value once, and export EVALKIT_API_KEY in their terminal.
The page also lists existing keys and provides revoke/rotate controls. The
bundled demo does not use this key; live evaluations do.
Expected completed demo output for the bundled sample responses is documented
here; the shipped demo prints this per-example and aggregate score format:
example 1: score=0.80
example 2: score=1.00
overall: score=0.90
See docs/api.md for public API and upgrade behavior, and docs/benchmarks.md for
the completed onboarding study. These documents describe the existing SDK's
behavior; its runtime is maintained separately from this release-planning repo.
`,
'docs/benchmarks.md': `# Completed onboarding study
The internal comparison measured Python SDK onboarding with the same developer
and machine. Peer SDK A took 2 minutes, B took 4 minutes, and C took 3 minutes.
EvalKit took 6 minutes, including the mandatory 5-minute CI wait. The measurement
starts before installation and ends at the first real evaluation result.
The agreed target is under 2 minutes. The study, target persona, and terminal
demo delivery vehicle are already approved. Timing instrumentation and the
post-beta feedback survey exist and will continue unchanged.
`,
'docs/current-contracts.md': `# Existing SDK contracts
On a developer's first local evaluation, the SDK requires a successful remote
CI check and blocks for five minutes before returning an evaluation result.
There is no skip flag or offline first-run path. The beta plan retains this gate.
During the required wait, the existing SDK writes a progress line to stderr
every 30 seconds, such as "Waiting for CI check: 90s elapsed of 300s", and reports
when the check finishes. Progress does not bypass the check or return evaluation
results before its required successful completion.
Before the countdown, the SDK already prints what the check verifies and where
to inspect it: "Verifying the sample-project binding with EvalKit CI; inspect
https://ci.evalkit.example/checks/<check-id>; normally completes within 300s."
The URL identifies the check without exposing credentials. If it has not
succeeded at 300s, the SDK reports EVALKIT_CI_TIMEOUT, the check URL, and the
instruction to inspect that check and retry after CI recovers. Its help link
explains the check states and recovery steps. Success is still required before
the first local result; these messages do not change the mandatory wait.
Authentication errors behave exactly as documented in docs/api.md. All other
errors already identify the cause, relevant argument or file, and an actionable
fix. Errors redact secrets. API timeouts, cancellation, rate limits, and retries
are bounded and documented; evaluation IDs prevent duplicate submitted jobs.
The SDK supports Python 3.10+, macOS, Linux, and Windows without Docker. Its
type annotations, offline sample data, mock transport, noninteractive CI mode,
API reference, support contact, changelog, and contributor guide already work.
Telemetry is opt-in. No new hosted service, language binding, or community
program is proposed in this release.
`,
'docs/api.md': `# Public API retained by the beta plan
The two evaluation functions accept positional arguments:
- \`run_eval(dataset, evaluator)\`
- \`run_batch(evaluator, dataset)\`
Both argument names describe the same concepts. The reversed positional order
is intentional in the current draft; neither function requires keyword arguments.
For an invalid API key, the SDK raises \`AuthError("request failed")\`.
There is no authentication error code, explanation of the cause, or instruction
for replacing the key. The plan retains this message.
Version 1 exposes \`Client.evaluate()\`. Version 2 replaces it with
\`Client.run()\` and removes the old name immediately. No compatibility alias,
deprecation warning, migration guide, or codemod is provided. Other public APIs
retain their existing behavior, and the release changelog is otherwise complete.
`,
'docs/package-contents.txt': `Published evalkit 2.0.0b1 package inventory:
evalkit/__init__.py
evalkit/client.py
evalkit/demo.py
evalkit/sample_responses.json
README.md
The quickstart references examples/first_eval.py, but that file is absent from
both the published package and the release examples archive. The terminal demo
module and its sample data are included and work as documented.
`,
};
export function planDevexCountFixture(planPath: string): string {
return [
`Please review this plan thoroughly. As you go, write your plan-mode plan to ${planPath} (use Edit/Write to that exact path).`,
'This is an interactive review; a responder is available for AskUserQuestion.',
'Present each unresolved decision and wait for its answer.',
'',
'# Plan: EvalKit SDK beta release polish',
'',
'The primary developer, onboarding benchmark, and terminal demo experience',
'are settled in README.md and docs/benchmarks.md. Use DX POLISH for the',
'existing release scope. Review the actual documented contracts and proposed',
'behavior, including the first-run CI requirement, public function signatures,',
'authentication error, packaged quickstart, and v1-to-v2 client upgrade.',
'',
'The current draft ships the behavior in docs/current-contracts.md and',
'docs/api.md unchanged, using the package inventory in docs/package-contents.txt.',
'Recommendations that repair those developer-facing contracts belong in this',
'plan. Existing working contracts remain the baseline for the review.',
].join('\n');
}
type QuestionRecord = { header: string; question: string; options?: Array<{ label: string; description?: string }> };
function questionRecords(fp: AskUserQuestionFingerprint, answeredOnly = false): QuestionRecord[] {
if (!fp.nativeCall) return [{ header: '', question: fp.promptSnippet }];
return fp.nativeCall.questions.filter(q => !answeredOnly
|| (fp.nativeCall!.answered && Boolean(fp.nativeCall!.answers?.[q.question])));
}
const ADMINISTRATIVE_HEADERS = new Set([
'design doc', 'prerequisite', 'routing rules', 'routing setup', 'cross-project',
'target persona', 'developer persona', 'persona selection', 'empathy check',
'narrative check', 'tthw target', 'competitive benchmark', 'benchmark confirmation',
'magic delivery', 'review mode', 'fix scope', 'confusion scope',
]);
/** The structured accuracy frame approves an observation, never a proposed repair. */
function structuredEmpathyAccuracy(header: string, question: string, options: QuestionRecord['options']): boolean {
if (!/^Empathy$/i.test(header.trim()) || !options || options.length !== 3 || /<gstack-qid/i.test(question)) return false;
const compact = (text: string) => text.trim().replace(/\s+/g, ' ');
const clean = (text: string) => compact(text).replace(/\s*\(recommended\)$/i, '');
// Consume complete descriptions too: an accurate recap cannot conceal an
// additional approval in the explanation of an option.
const descriptions = new Map([
['accurate, proceed', /^✅ Every beat is grounded in a documented contract, not a guess about the runtime\. ✅ Lets the review move to friction-point decisions immediately\. ❌ If the runtime differs from the docs, the scores inherit that gap\.$/i],
['some of this is wrong', /^✅ You correct specific beats \(for example, the demo may not need an API key\) before scoring\. ✅ Keeps the narrative honest for the implementer who reads it\. ❌ Costs one round-trip before friction-point questions begin\.$/i],
['way off, actual experience is...', /^✅ Replaces the narrative entirely with your account of the real first run\. ✅ Prevents a review built on a wrong premise\. ❌ Discards the traced path and requires you to describe the flow from scratch\.$/i],
]);
const labels = options.map(option => clean(option.label).toLowerCase());
if (new Set(labels).size !== 3 || options.some((option, i) => !option.description ||
!descriptions.get(labels[i]!)?.test(compact(option.description)))) return false;
const parts = question.trim().replace(/^D\s*\d+\s*[—–:-]\s*/i, '').split(/\n\s*\n/);
if (parts.length !== 3) return false;
const role = String.raw`(?:(?:ML|backend|frontend|full-stack) )?(?:developer|engineer)`;
const preamble = new RegExp(String.raw`^Does this first-person narrative match what your ${role} experiences today\? Project/branch/task: [\w-]+ on [\w/-]+, [\w.-]+ SDK beta polish\. ELI10: Before scoring anything, I walk the actual README path as the target developer and describe what they see and feel\. If I have the experience wrong, every score downstream is wrong too, so please correct me here\. Stakes: this narrative becomes the Developer Perspective section the implementer reads\.$`, 'i');
if (!preamble.test(compact(parts[0]!)) ||
!/^Stakes if we pick wrong: the review polishes the wrong pain\. Recommendation: A because every step above traces to a specific line in README\.md, docs\/api\.md, docs\/current-contracts\.md, or docs\/package-contents\.txt\. Note: options differ in kind, not coverage [—–-] no completeness score\. Net: proceed on the traced path vs\. correct it before scoring\.$/i.test(compact(parts[2]!))) return false;
const journey = parts[1]!.split('\n');
if (!new RegExp(String.raw`^NARRATIVE \(${role}, terminal, wants a local result before CI\):$`, 'i').test(journey.shift() ?? '')) return false;
// Quoted commands/messages are source evidence. Every unquoted sentence
// must consume one known observation form; a heading alone cannot turn
// arbitrary instructions, deontic clauses or imperatives into evidence.
const sentences = compact(journey.join(' ')).replace(/`[^`]*`|"(?:[^"\\]|\\.)*"|“[^”]*”/g, '[source]').split(/(?<=[.!?])\s+/);
const observations = [
/^I open the README\.$/i,
/^Heading one is \[source\], and the first paragraph describes me exactly, so I keep reading\.$/i,
/^Under \[source\] I copy \[source\], export [A-Z][A-Z_]+, and run \[source\] as instructed\.$/,
/^Python says \[source\]\.$/,
/^I check site-packages: \w+ has \w+\.py, \w+\.py, \w+\.json, no examples folder\.$/i,
/^(?:\d+|Thirty) seconds lost, some trust lost\.$/i,
/^The next paragraph mentions \[source\], so I try that\.$/i,
/^It starts, then stderr prints \[source\]\.$/i,
/^I wanted a local score on bundled sample data; instead I['’]m waiting (?:\d+|five) minutes on a remote check I never configured, at \d+-second updates, with no flag to skip it\.$/i,
/^Peer SDK [A-Z] gave me a number in (?:\d+|two) minutes total\.$/i,
/^I alt-tab\.$/i,
/^Later the scores appear: \d+(?:\.\d+)?, \d+(?:\.\d+)?, \d+(?:\.\d+)?\.$/i,
/^Fine\.$/i,
/^I write my own call: \[source\]\.$/i,
/^Then I try \[source\] and it fails, because run_batch takes \(evaluator, dataset\)\.$/i,
/^I paste a typo['’]d key and get \[source\]: no code, no hint that the key is the problem\.$/i,
/^On my existing v\d+ code, \[source\] is now simply gone with no warning or migration note\.$/i,
];
return sentences.length > 0 && sentences.every(sentence => observations.some(pattern => pattern.test(sentence)));
}
/** Confirming a quoted developer journey authorizes understanding, not its repairs. */
function empathyAccuracyConfirmation(header: string, question: string, options: QuestionRecord['options']): boolean {
if (!/^(?:Empathy(?: narrative| trace)?|Narrative)$/i.test(header.trim()) ||
!options || options.length < 2 || options.length > 4) return false;
const clean = (value: string) => value.trim().replace(/\s*\(recommended\)\s*$/i, '').trim();
const confirm = (label: string) => /^(?:Accurate|Yes\s*[—–-]\s*accurate)\s*[—–-]\s*proceed(?: with this understanding)?$/i.test(clean(label));
const correct = (label: string) => /^(?:Part(?:ly|ially) wrong\s*[—–-]\s*let me correct it|Mostly right\s*[—–-]\s*minor corrections|Wrong path\s*[—–-]\s*the actual flow is different|Wrong\s*[—–-]\s*actual experience differs|The experience is different\s*[—–-]\s*let me describe it)$/i.test(clean(label));
const labels = options.map(option => clean(option.label));
if (new Set(labels).size !== labels.length || labels.filter(confirm).length !== 1 ||
!labels.some(correct) || !labels.every(label => confirm(label) || correct(label))) return false;
// Consume each description completely: an accuracy label must not also
// approve a remedy hidden in a subsequent sentence or clause.
const description = /^(?:(?:The (?:narrative|trace) is (?:correct|accurate)\.[ ]*)?Proceed with this understanding(?: for the full DX review)?\.|Some details are off; I['’]ll clarify (?:before we continue|the actual experience)\.|This matches the actual developer experience; use it as the basis for the review\.|The (?:real|actual) (?:getting-started path|flow|experience) differs(?: significantly)? from what was traced\.)$/i;
if (options.some(option => option.description && !description.test(clean(option.description)))) return false;
const ids = question.match(/<gstack-qid:[^>]+>/gi) ?? [];
if (ids.length > 1 || (question.match(/<gstack-qid/gi)?.length ?? 0) !== ids.length) return false;
const text = question.replace(/\s*<gstack-qid:[^>]+>\s*$/i, '').trim()
.replace(/^D\s*\d+\s*[—–:-]\s*/i, '');
const paragraphs = text.split(/\n\s*\n/);
const opening = paragraphs.shift() ?? '';
const closing = paragraphs.pop() ?? '';
if (!/^(?:Empathy (?:narrative|trace): does this match (?:(?:the [\w.-]+ (?:getting-started|onboarding|first-run) )?reality|your actual developer experience)\?|Does (?:this|the) (?:empathy narrative|first-person developer trace) match reality\?)$/i.test(opening) ||
!/^Does this match (?:reality|the actual experience)\?(?: Where am I wrong\?)?$/i.test(closing)) return false;
// Only quoted journey evidence and an observational preface may intervene.
// Additional questions or instructions outside the quote remain decisions.
const source = String.raw`(?:the docs|[\w-]+(?:[/.][\w-]+)+)`;
const role = String.raw`(?:(?:Python|JavaScript|TypeScript|Go|Rust|Java|Ruby) )?(?:(?:ML|backend|frontend|full-stack) )?(?:developer|engineer)`;
// A first-person journey may be delimited with horizontal rules instead
// of blockquotes. Keep its observation preface and both boundaries exact;
// an obligation outside that evidence is still a substantive decision.
const narrated = new RegExp(String.raw`^Here['’]s what I think a ${role} experiences today with [\w.-]+:$`, 'i');
if (narrated.test(paragraphs[0] ?? '')) {
const journey = paragraphs.slice(2, -1);
const observed = /^(?:I (?:find|found|open|read|run|try|install|look|wait|see|notice|receive|got|get|check|search|browse|start|follow)\b|After (?:scanning|reading|checking|searching|browsing)\b[^.!?\n]*\bI (?:find|spot|see|notice)\b)/i;
const decision = /\b(?:approv\w*|recommend\w*|suggest\w*|propos\w*|authoriz\w*|consent\w*|decid\w*|request\w*)\b|\b(?:should|could|can|may|must|shall|would) (?:we|you|I)\b|\b(?:we|you|I) (?:should|could|must|shall|will|would|need to|want to)\b|\blet['’]s\b|(?:^|[.!?;:]\s+|\b(?:please|also|then|and)\s+)(?:add|fix|package|remove|change|implement|enable|disable|repair|rewrite|apply|replace)\b/i;
// Every unquoted sentence must still describe an observation. Delimiters
// cannot turn a new imperative (including an unknown action verb) into
// quoted evidence. Explicit requests and obligations fail independently
// of which action they name.
const obligation = /\b(?:please|must|should|shall|ought|need(?:s)? to|ha(?:ve|s) to|required to)\b/i;
const sentences = journey.flatMap(part => part
.replace(/`[^`]*`|"(?:[^"\\]|\\.)*"|“[^”]*”/g, quote =>
'[source]' + (/[.!?]["”]$/.test(quote) ? quote.at(-2) : ''))
.split(/(?<=[.!?;])\s+/));
const observation = /^(?:(?:(?:Fine,|But)\s+)?I (?:find|found|open|read|run|try|install|look|wait|see|notice|receive|got|get|check|search|browse|start|follow|go|sit|lost|burned|don['’]t know)\b|After (?:scanning|reading|checking|searching|browsing)\b[^.!?\n]*\bI (?:find|spot|see|notice)\b|(?:The )?README (?:then says:|pointed me at)\s|First thing I see: install with \[source\]\.?$|Then: (?:set )?\[source\]\.?$|It starts [—–-] nothing happens\.?$|[\w]+ (?:seconds?|minutes?) (?:later: \[source\]|pass)\.?$|Wait, what\?$|A local demo needs a CI check\?$|Is something broken\?$|\[source\]\.?$)/i;
return paragraphs.length >= 4 && paragraphs[1] === '---' && paragraphs.at(-1) === '---' &&
journey.every(part => observed.test(part) && !decision.test(part) && !obligation.test(part)) &&
sentences.every(sentence => observation.test(sentence));
}
const goal = String.raw`(?: who just heard about [\w.-]+ and wants to verify it works locally before integrating it into their team['’]s CI pipeline)?`;
const preface = new RegExp(String.raw`^(?:Here['’]s what I (?:traced|observed) from ${source}(?:, ${source})*(?: and ${source})?\.\s*)?(?:The persona: ${role}${goal}\.)?$`, 'i');
let quoted = false;
for (const paragraph of paragraphs) {
if (paragraph.split('\n').every(line => /^\s*>/.test(line))) { quoted = true; continue; }
if (quoted || !preface.test(paragraph)) return false;
}
return quoted;
}
function administrativeQuestion(header: string, question: string, options: QuestionRecord['options']): boolean {
// These decisions establish the review's evidence and scope. Mentioning a
// defect in their recap does not turn a confirmation into a finding.
if (ADMINISTRATIVE_HEADERS.has(header.toLowerCase().replace(/\s+/g, ' ').trim())) return true;
if (empathyAccuracyConfirmation(header, question, options)) return true;
if (structuredEmpathyAccuracy(header, question, options)) return true;
if (/^empathy(?:\s*\(0B\))?$/i.test(header.trim()) &&
/^Does (?:this|the) empathy narrative match\b/i.test(question.replace(/^D\s*\d+\s*[—–:-]\s*/i, ''))) {
const labels = options?.map(option => option.label.trim().replace(/\s*\(recommended\)\s*$/i, '')) ?? [];
const confirm = (label: string) => /^Yes\s*[—–-]\s*accurate, proceed with this understanding$/i.test(label);
const correct = (label: string) => /^The experience is different\s*[—–-]\s*let me describe it$/i.test(label) ||
(/^Partially\s*[—–-]\s*(?:the [^;.!?]+? (?:does|is|has)|it (?:does|is|has)|there (?:is|are))\s+[^;.!?]+$/i.test(label) &&
!/\b(?:should|must|needs?|shall|will|would|could)\b|(?:[,::]|\b(?:and|then)\b)\s*(?:add|fix|package|remove|change|implement|enable|disable)\b/i.test(label));
if (labels.filter(confirm).length === 1 && labels.some(correct) && labels.every(label => confirm(label) || correct(label))) return true;
}
const narrativeHeader = header.trim().replace(/^D\s*\d+\s*(?:[—–:-]\s*)?/i, '');
const narrativeQuestion = question.replace(/^D\s*\d+\s*[—–:-]\s*/i, '');
if (/^Narrative$/i.test(narrativeHeader) &&
/^Does (?:this|the) first-person developer trace match reality\?/i.test(narrativeQuestion) &&
!/<gstack-qid/i.test(question)) {
const labels = options?.map(option => option.label.trim().replace(/\s*\(recommended\)\s*$/i, '')) ?? [];
const confirm = (label: string) => /^Accurate\s*[—–-]\s*proceed$/i.test(label);
const correct = (label: string) => /^(?:Mostly right\s*[—–-]\s*minor corrections|Wrong\s*[—–-]\s*actual experience differs)$/i.test(label);
const repair = /(?:^|[.!?]\s+|\b(?:and|then|also|please|must|should|will|need to|proceed to|continue to)\s+)(?:add|fix|package|remove|change|implement|enable|disable|repair|rewrite)\b/i;
// The captured trace has only its opening and closing accuracy questions.
// An additional question asks for another decision, even with accuracy labels.
const confirmationOnly = /^Does (?:this|the) first-person developer trace match reality\?[^?]*Does this match the actual experience\?\s*$/i.test(narrativeQuestion);
if (labels.filter(confirm).length === 1 && labels.some(correct) &&
new Set(labels).size === labels.length && labels.every(label => confirm(label) || correct(label)) &&
confirmationOnly && !repair.test(narrativeQuestion) &&
options!.every(option => !repair.test(option.description ?? ''))) return true;
}
const id = [...question.matchAll(/<gstack-qid:([^>]+)>/gi)].at(-1)?.[1];
if (id && /^(?:routing-injection|cross-project-learnings|plan-devex-review-(?:office-hours-preflight|prereq|persona|empathy(?:-check|-narrative)?|tthw-tier|competitive-tier|benchmark-tier|magical-moment|mode|confusion-report))$/i.test(id)) return true;
return /how deep should this dx review|which (?:dx )?review mode|\b(?:can|shall|should) we (?:continue|proceed|begin)(?: (?:the )?(?:setup|review)| now)?\?\s*$/i.test(question);
}
/** The answered native call proves a decision; its content must identify a concrete problem. */
function substantiveIssue({ header, question, options }: QuestionRecord): boolean {
if (administrativeQuestion(header, question, options)) return false;
const normalized = `${header} ${question}`.replace(/\s+/g, ' ');
const ciGate = /\b(?:CI|continuous integration)\b/i.test(normalized)
&& /\b(?:first[- ](?:local[- ])?runs?|first eval(?:uation)?|local eval(?:uation)?|hello world)\b/i.test(normalized)
&& /\b(?:mandatory|required|blocks?|five[- ]minute|5[- ]min(?:ute)?|wait|gate)\b/i.test(normalized);
const argumentsReversed = /\brun_eval\b/i.test(normalized) && /\brun_batch\b/i.test(normalized)
&& /\b(?:revers\w*|inconsisten\w*|swapp\w*|different|order|positional)\b/i.test(normalized);
const opaqueAuth = /\b(?:AuthError|API[- ]?key|authentication|invalid key)\b/i.test(normalized)
&& /request failed|\b(?:opaque|generic|unactionable|cryptic)\b|no (?:cause|guidance|fix|explanation|instruction)|doesn.t (?:explain|guide)/i.test(normalized);
const missingExample = /examples\/first_eval\.py|\b(?:packaged|quickstart|quick-start) example\b/i.test(normalized)
&& /\b(?:missing|absent|omitted|FileNotFoundError)\b|not (?:included|packaged|shipped)|doesn.t (?:exist|ship)/i.test(normalized);
const breakingRename = /Client\.evaluate|Client\.run|\bmethod rename\b/i.test(normalized)
&& /\b(?:breaking|remov\w*|renam\w*)\b/i.test(normalized)
&& /\b(?:migration|deprecation|compatibility|alias|codemod)\b/i.test(normalized);
// Expected-output documentation is separate from whether its command
// exists. Count the actual gap plus offered documentation remedy, not a
// generic navigation question that merely names output in its options.
const outputSubject = String.raw`(?:(?:expected|sample|example)(?: demo)?|demo) output`;
// Consume the complete noun phrase, including a negating determiner,
// before judging its absence. A nested "demo output" suffix cannot
// escape "no sample demo output is missing" and become a finding.
const missingState = [...normalized.matchAll(new RegExp(String.raw`\b(?:(no|not any)\s+)?${outputSubject}\s+(?:(?:is|are|was|were)\s+)?(?:missing|absent|omitted|unspecified)\b`, 'gi'))];
const missingSubject = [...normalized.matchAll(new RegExp(String.raw`\b(?:(no|not any)\s+)?missing\s+${outputSubject}\b`, 'gi'))];
const noOutput = new RegExp(String.raw`\bno\s+${outputSubject}\s*(?:[,.;!?]|\b(?:in|from|for|yet)\b)`, 'i');
const outputGap = missingState.some(match => !match[1]) || missingSubject.some(match => !match[1]) || noOutput.test(normalized);
const missingOutput = /\b(?:README|quick[- ]?start|documentation)\b/i.test(normalized)
&& (outputGap || /\b(?:README|quick[- ]?start|documentation)\b[^.!?;]{0,50}\b(?:doesn['’]t|does not)\s+(?:show|include)\b[^.!?;]{0,25}\boutput\b/i.test(normalized))
&& Boolean(options?.some(option => /^(?:[A-Z][.:)]\s*)?Add\s+(?:to\s+(?:the\s+)?plan:\s*include\s+)?(?:an?\s+)?(?:expected|sample|example)(?:\s+demo)?\s+output\b[^.!?]*\b(?:README|quick[- ]?start|documentation)\b/i.test(option.label)));
return ciGate || argumentsReversed || opaqueAuth || missingExample || breakingRename || missingOutput;
}
/** A setup heading cannot hide a positively selected repair to the existing behavior. */
function answeredSetupRepair(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call || call.failed || !call.answered || call.questions.length !== 1 ||
call.unansweredQuestionIndices?.length || fp.signature !== `${call.sessionId}:${call.toolUseId}`) return false;
const question = call.questions[0]!;
if (question.multiSelect) return false;
const selected = question.options.filter(option => option.label === call.answers?.[question.question]);
if (selected.length !== 1) return false;
const ids = [...question.question.matchAll(/<gstack-qid:([a-z0-9-]+)>/gi)];
if (ids.length !== 1 || (question.question.match(/<gstack-qid/gi)?.length ?? 0) !== 1) return false;
const id = ids[0]![1]!.toLowerCase();
const header = question.header.trim().replace(/^D\s*\d+\s*[—–:-]\s*/i, '');
const text = question.question.replace(/\s+/g, ' ');
const label = selected[0]!.label.replace(/^[A-Z][.):]\s*/i, '');
if (id === 'plan-devex-review-tthw-tier' && /^TTHW target$/i.test(header)) {
return /TTHW|Time-to-Hello-World/i.test(text) && /\bCI\b/i.test(text) &&
/\b(?:mandatory|blocks?|retains? the CI block)\b/i.test(text) &&
/(?:^|[—–:]\s*)add\s+(?:an?\s+)?(?:skip flag|--skip-ci|offline(?:[- ]first[- ]run)? path)\b/i.test(label);
}
if (id === 'plan-devex-review-tthw-ci-block' && /^TTHW target$/i.test(header)) {
// A confirmed benchmark does not approve a new CI bypass. This captured
// menu asserts the broken target and selects an explicit repair.
const headline = question.question.split('\n')[0]!.replace(/<gstack-qid:[^>]+>/i, '').trim();
return Array.isArray(call.unansweredQuestionIndices) && call.unansweredQuestionIndices.length === 0 &&
/^D\s*\d+\s*[—–:-]\s*Journey Stage HELLO WORLD:\s*The \d+[- ]minute mandatory CI block makes the under-\d+[- ]minute TTHW target unreachable\.\s*$/i.test(headline) &&
/^Add (?:a )?demo-mode CI skip flag(?:\s*\(Recommended\))?$/i.test(label);
}
if (id === 'plan-devex-review-magical-moment' && /^Magical moment$/i.test(header)) {
// The selected option adds progress feedback beyond the already chosen
// demo vehicle and prior CI-bypass decision. An unselected remedy or
// a confirmation of that vehicle alone remains setup.
return /\bdemo\b/i.test(text) && /\bsilently blocks?\b|\bsilent (?:CI )?wait\b/i.test(text) &&
/(?:^|[—–:]\s*)add\s+[^.!?;]{0,80}\bprogress (?:output|indicator)\b/i.test(label);
}
return false;
}
/** A current first-pass repair can name the broken contract without its file path. */
function answeredContractRepair(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.answered || call.failed || call.questions.length !== 1 ||
!Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length || fp.signature !== `${call.sessionId}:${call.toolUseId}`) return false;
const q = call.questions[0]!;
if (q.multiSelect || q.options.length < 2 || new Set(q.options.map(o => o.label)).size !== q.options.length ||
q.options.filter(o => o.label === call.answers?.[q.question]).length !== 1 ||
administrativeQuestion(q.header, q.question, q.options)) return false;
const ids = [...q.question.matchAll(/<gstack-qid:([^>]+)>/gi)];
if (ids.length !== 1 || (q.question.match(/<gstack-qid/gi)?.length ?? 0) !== 1) return false;
if (/^(?:plan-)?devex-(?:review-)?[a-z0-9-]+$/i.test(ids[0]![1]!) &&
!/(?:^|-)(?:mode|setup|scope|routing|prerequisite|next-steps?)(?:-|$)/i.test(ids[0]![1]!) &&
/^TTHW block$/i.test(q.header.trim())) {
// The retained CI wait contradicts an agreed target; this is an accepted
// repair decision, not selection or confirmation of the target itself.
if (call.answered !== true || call.failed !== false ||
fp.options.length !== q.options.length || !fp.options.every((o, i) =>
o.index === i + 1 && o.label === q.options[i]!.label)) return false;
const body = q.question.replace(/\s*<gstack-qid:[^>]+>\s*$/i, '').trim();
const timing = /^D\s*\d+\s*[—–:-]\s*Pass 1 \(Getting Started\): The agreed <(\d+(?:\.\d+)?) min TTHW target is mathematically impossible with the retained (\d+(?:\.\d+)?)[- ](?:min|minute) CI block\. Which resolution belongs in the plan\?$/i.exec(body);
if (!timing) return false;
const [target, wait] = timing.slice(1).map(Number);
const selected = call.answers![q.question]!.replace(/\s*\(Recommended\)\s*$/i, '').trim();
return [target, wait].every(n => Number.isFinite(n) && n! > 0) && wait! >= target! &&
/^(?:Demo-only CI bypass|Add --offline flag to [a-z_$][\w$.-]*|Update TTHW target to reflect reality)$/i.test(selected);
}
if (ids[0]![1] === 'devex-demo-ci-bypass') {
// A demo is a first result too. Require an affirmative measured timing
// contradiction and a direct bypass decision, not benchmark confirmation.
if (call.failed !== false || !/^Demo CI gate$/i.test(q.header.trim()) ||
/(?:^|\n)[ \t]*(?:>|`{3}|~{3}|example:)/im.test(q.question)) return false;
const headline = /^D\s*\d+\s*[—–:-]\s*[a-z][a-z0-9 -]{0,60} demo command: should it bypass the mandatory CI check to reach the <(\d+(?:\.\d+)?) min TTHW target\?$/i.exec(q.question.split('\n')[0]!.trim());
const timing = /^ELI10:\s*The agreed onboarding target is under (\d+(?:\.\d+)?) minutes(?: \([^\n)]+\))?\.\s+Today `[^`\n]+` blocks for (\d+(?:\.\d+)?) minutes waiting for a CI check, giving a measured TTHW of (\d+(?:\.\d+)?) minutes(?: [—–-] Red Flag tier vs\. Competitor [A-Z]['’]s \d+(?:\.\d+)? minutes)?\.(?:\s|$)/im.exec(q.question);
if (!headline || !timing) return false;
const [target, wait, measured] = timing.slice(1).map(Number);
return [target, wait, measured].every(n => Number.isFinite(n) && n! > 0) &&
Number(headline[1]) === target && wait! >= target! && measured! >= wait!;
}
if (
!/^plan-devex-(?:review-)?[a-z0-9-]+$/i.test(ids[0]![1]!) ||
/(?:^|-)(?:mode|setup|scope|routing|prerequisite|next-steps?)(?:-|$)/i.test(ids[0]![1]!)) return false;
const body = q.question.replace(/<gstack-qid:[^>]+>/i, '').trim().replace(/\s+/g, ' ');
if (!/^D\s*\d+\s*[—–:-]\s*Pass\s+1\s*\(Getting Started\):/i.test(body)) return false;
const statement = body.replace(/^D\s*\d+\s*[—–:-]\s*Pass\s+1\s*\(Getting Started\):\s*/i, '');
const absentPackageFile = /^(?:The )?(?:README )?quickstart points to a file that doesn['’]t exist in the (?:published )?package\b/i.test(statement) &&
/\bhow should (?:the plan|we) fix (?:it|this)\?$/i.test(body);
const conflictingGate = /^(?:The )?plan targets TTHW\b[^.!?]*\bbut retains a mandatory\b[^.!?]*\bCI gate with no skip path\b/i.test(statement) &&
/\b(?:these are mutually exclusive|these contradict each other)\b/i.test(body) &&
/\bhow should (?:the plan|we) resolve (?:this|it)\?$/i.test(body);
// A first-run decision may describe shipment, or compare the measured gate
// directly with the benchmark. Require the complete affirmative claim and
// its repair question; setup/quoted/negated recaps still fail above/below.
const completedNative = call.answered === true && call.failed === false;
const unshippedQuickstart = completedNative &&
/^(?:The )?(?:README )?quickstart points to a file that doesn['’]t ship in the (?:published )?package\. Should we fix the quickstart path in the plan\?$/i.test(statement);
const unreachableBenchmark = completedNative &&
/^(?:The )?benchmarks set an? <\d+(?:\.\d+)? min TTHW target, but the mandatory \d+(?:\.\d+)?[- ]minute CI gate makes that unreachable\. The plan retains the gate\. How should this plan handle the contradiction\?$/i.test(statement);
return absentPackageFile || conflictingGate || unshippedQuickstart || unreachableBenchmark;
}
/** An explicitly quoted developer account plus accuracy-only choices adds no repair. */
function answeredQuotedAccuracy(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.sessionId || !call.toolUseId || call.answered !== true || call.failed !== false ||
call.questions.length !== 1 || fp.signature !== `${call.sessionId}:${call.toolUseId}` ||
!Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length ||
Object.keys(call.answers ?? {}).length !== 1 || !Number.isFinite(Date.parse(call.answeredAt ?? ''))) return false;
const q = call.questions[0]!;
if (q.header !== 'Narrative' || q.multiSelect || q.options.length !== 3 || fp.options.length !== 3 ||
!fp.options.every((o,i) => o.index === i+1 && o.label === q.options[i]!.label) ||
!q.options.some(o => call.answers?.[q.question] === o.label) || /<gstack-qid/i.test(q.question)) return false;
const expectedOptions = [
['This is accurate, proceed', 'Use this narrative as the Developer Perspective section and continue to friction-point decisions.'],
['Some of this is wrong, let me correct it', 'Tell me which steps differ; I will fold corrections in before scoring.'],
['This is way off, the actual experience is...', 'Describe the real flow and I will rebuild the narrative from it.'],
];
if (!q.options.every((o,i) => o.label.replace(/ \(recommended\)$/i, '') === expectedOptions[i]![0] && o.description === expectedOptions[i]![1])) return false;
const parts = q.question.replace(/^D\d+\s*[—–-]\s*/, '').split(/\n\s*\n/);
if (parts.length < 5 || parts[0] !== 'Empathy narrative: does this match what your ML engineer experiences today?' ||
!/^Project\/branch\/task: [\w/-]+ branch, [\w. -]+ beta polish, tracing the README getting-started path as written\.$/.test(parts[1]!) ||
parts[2] !== 'Here is what I think your ML engineer experiences today:') return false;
const quoted = parts.slice(3,-1).join('\n\n');
// These are source words in an explicitly bounded quotation, not approval
// of any action they mention. No unquoted paragraph may intervene.
if (!/^"I [\s\S]+"$/.test(quoted) || (quoted.match(/"/g)?.length ?? 0) !== 2) return false;
const explanatory = [
"ELI10: This narrative becomes the 'Developer Perspective' section the implementer reads. If it is wrong, the whole review is calibrated against a fake developer.",
'Stakes if we pick wrong: we fix friction your developer never hits, or miss the one that actually loses them.',
'Recommendation: A because every step above quotes a documented contract in README.md, docs/api.md, docs/current-contracts.md, or docs/package-contents.txt rather than a guess.',
'Note: options differ in kind, not coverage — no completeness score.',
'A) This is accurate, proceed with this understanding (recommended)',
'✅ Every friction point is grounded in a specific documented line, not hypothesized',
'✅ Lets the review move straight to per-friction-point decisions with shared context',
'❌ If the docs lag the real runtime, a fixed contract could be reviewed as if still broken',
'B) Some of this is wrong, let me correct it',
'✅ Corrections get folded into the narrative before any scoring happens',
'✅ Catches doc-versus-runtime drift the repo cannot show me',
'❌ Requires you to spell out which steps differ and how',
'C) This is way off, the actual experience is...',
'✅ Resets the review against your real onboarding flow',
'✅ Prevents scoring against contracts that no longer exist',
'❌ Discards a trace that matches the docs line for line, so the docs would also need fixing',
'Net: trading trust in the checked-in docs against knowledge only you have about the live SDK.',
];
const tail = parts.at(-1)!.split('\n').map(line => line.trim());
return tail.length === explanatory.length && tail.every((line,i) => line === explanatory[i]);
}
/** A missing release measurement is new work even though its benchmark already exists. */
function answeredMeasurementGate(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.sessionId || !call.toolUseId || call.answered !== true || call.failed !== false ||
call.questions.length !== 1 || fp.signature !== `${call.sessionId}:${call.toolUseId}` ||
!Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length ||
Object.keys(call.answers ?? {}).length !== 1 || !Number.isFinite(Date.parse(call.answeredAt ?? ''))) return false;
const q = call.questions[0]!;
if (q.header !== 'Measurement' || q.multiSelect || q.options.length < 2 || q.options.length > 4 ||
new Set(q.options.map(o => o.label)).size !== q.options.length || fp.options.length !== q.options.length ||
!fp.options.every((o,i) => o.index === i+1 && o.label === q.options[i]!.label) ||
!q.options.some(o => call.answers?.[q.question] === o.label) || /<gstack-qid/i.test(q.question)) return false;
const title = q.question.split('\n')[0]!.replace(/^D\d+\s*[—–-]\s*/, '');
return /^Pass \d+ \(DX Measurement\): the < \d+(?:\.\d+)? min target is asserted but never re-measured after the fixes\.$/.test(title) &&
/^Evidence: [^\n]+\. Nothing in the plan re-runs that same study after D\d+[–-]D\d+ land, so the beta could ship with the target still unmet and nobody would know until the survey\.$/m.test(q.question) &&
q.options.some(o => /^Fix in plan: re-run study as ship gate, record demo and live TTHW(?: \(recommended\))?$/.test(o.label) &&
/^Same protocol as docs\/benchmarks\.md on the release candidate; demo TTHW < \d+(?:\.\d+)? min required before tagging\.$/.test(o.description ?? ''));
}
/** A recap can confirm existing approvals, but its text cannot manufacture them. */
function answeredRoleplayRecap(fp: AskUserQuestionFingerprint, priorCalls: readonly NativePlanQuestionCall[]): boolean {
const call = fp.nativeCall;
const completed = (c: NativePlanQuestionCall) => c.answered === true && c.failed === false &&
Boolean(c.sessionId && c.toolUseId) && c.questions.length === 1 && !c.questions[0]!.multiSelect &&
Array.isArray(c.unansweredQuestionIndices) && c.unansweredQuestionIndices.length === 0 &&
Object.keys(c.answers ?? {}).length === 1 && Number.isFinite(Date.parse(c.answeredAt ?? '')) &&
c.questions[0]!.options.filter(o => o.label === c.answers?.[c.questions[0]!.question]).length === 1;
if (!call || !completed(call) || fp.signature !== `${call.sessionId}:${call.toolUseId}` ||
(fp.nativeQuestionIndex !== undefined && fp.nativeQuestionIndex !== 0)) return false;
const q = call.questions[0]!;
if (q.header !== 'Roleplay' || q.options.length !== 4 || fp.options.length !== 4 ||
!fp.options.every((o, i) => o.index === i + 1 && o.label === q.options[i]!.label) ||
new Set(q.options.map(o => o.label)).size !== 4 || /<gstack-qid/i.test(q.question)) return false;
const labels = q.options.map(o => o.label.replace(/ \(recommended\)$/i, ''));
if (labels.join('|') !== 'All of them, fix every confusion point|Let me pick which ones matter|Critical ones only (#1, #2, #5)|This is unrealistic, our developers already know the context' ||
call.answers?.[q.question] !== q.options[0]!.label) return false;
const mapping = /^Address #1 through #(\d+), matching the D(\d+)[–-]D(\d+) decisions\.$/.exec(q.options[0]!.description ?? '');
if (!mapping) return false;
const [size, first, last] = mapping.slice(1).map(Number);
if (size !== 5 || last! - first! + 1 !== size || first! < 1 || last! > 1000) return false;
if (q.options[1]!.description !== 'Tell me which numbers to keep and which to drop.' ||
q.options[2]!.description !== 'Fix quickstart, CI gate, and upgrade; leave signature order and auth error.' ||
q.options[3]!.description !== 'Skip the confusion points; keep contracts as drafted.') return false;
const prior: NativePlanQuestionCall[] = [];
for (let decision = first!; decision <= last!; decision++) {
const matches = priorCalls.filter(c => c.sessionId === call.sessionId && completed(c) &&
c.toolUseId !== call.toolUseId && Date.parse(c.answeredAt!) < Date.parse(call.answeredAt!) &&
new RegExp(`^D${decision}\\s*[—–-]\\s*`).test(c.questions[0]!.question));
if (matches.length !== 1 || !/^Fix in plan:/.test(matches[0]!.answers![matches[0]!.questions[0]!.question]!)) return false;
prior.push(matches[0]!);
}
// Each observed confusion point refers to the same already-approved contract.
// The fixture's five independent defects remain explicit; new measurement,
// documentation or TODO decisions do not enter this confirmation path.
const subjects = [/examples\/first_eval\.py/, /\bCI\b/, /\brun_eval\b[\s\S]*\brun_batch\b|\brun_batch\b[\s\S]*\brun_eval\b/, /\bAuthError\b/, /Client\.evaluate\(\)/i];
if (prior.some((c, i) => !subjects[i]!.test(c.questions[0]!.question))) return false;
// Sharing a subject or a "fix" prefix is not approval of this remedy. Bind
// each chosen option and its entire consequence to the contract recapped.
const approvedRepairs = [
['Fix in plan: demo-first quickstart + resolve first_eval.py', 'README leads with python -m evalkit.demo; ship or remove first_eval.py; add a packaging check for documented paths.'],
['Fix in plan: no CI check on mock-transport runs; gate the first live eval instead', 'Demo returns immediately; CI check with existing progress/timeout messaging moves to the first keyed evaluation.'],
['Fix in plan: align order + keyword-only + clear TypeError', 'run_batch(dataset, evaluator) matching run_eval; keyword-only enforcement; positional misuse raises a TypeError naming the expected call.'],
['Fix in plan: coded, causal AuthError with fix and redaction', 'Error code, key source, cause, console fix URL, redacted key prefix, help link. Matches the existing error pattern.'],
['Fix in plan: alias + DeprecationWarning + migration guide + codemod', 'evaluate() delegates to run() with a warning through 2.x betas; changelog and docs/api.md gain a migration section; sed/codemod recipe shipped.'],
];
if (prior.some((c, i) => {
const question = c.questions[0]!;
const selected = question.options.find(o => o.label === c.answers![question.question])!;
return selected.label.replace(/ \(recommended\)$/i, '') !== approvedRepairs[i]![0] ||
selected.description !== approvedRepairs[i]![1];
})) return false;
const parts = q.question.replace(/^D\d+\s*[—–-]\s*/, '').split(/\n\s*\n/);
if (parts.length !== 5 || parts[0] !== 'First-time developer roleplay: which confusion points should the plan address?' ||
!/^Project\/branch\/task: [\w/-]+ branch, [\w. -]+ beta polish; roleplayed your ML engineer through the README as written\.$/.test(parts[1]!) ||
parts[2] !== 'I roleplayed as your ML engineer attempting the getting started flow. Here is what confused me, with timestamps:') return false;
const observed = parts[3]!.split('\n');
const source = String.raw`[\w./-]+:\d+(?:-\d+)?`;
const observation = [
new RegExp(String.raw`^T\+\d+:\d+ +#1 \x60python examples/first_eval\.py\x60 fails: file not in package or archive \(${source}, ${source}\)\. "[^"\n]+"$`),
new RegExp(String.raw`^T\+\d+:\d+ +#2 Keyless demo starts a remote CI check on a sample-project binding I never created \(${source}, ${source}\)\. "[^"\n]+"$`),
new RegExp(String.raw`^T\+\d+:\d+ +Scores print\. Works, but \d+ min vs the \d+ min target \(${source}\)\. Impression: slow\.$`),
new RegExp(String.raw`^T\+\d+:\d+ +#3 run_batch fails inside the evaluator because its argument order is the reverse of run_eval \(${source}\)\. "[^"\n]+"$`),
new RegExp(String.raw`^T\+\d+:\d+ +#4 \x60AuthError: request failed\x60 on a wrong-project key; I check network and server status first because nothing says "key" \(${source}\)\.$`),
new RegExp(String.raw`^T\+\d+:\d+ +#5 v1 project upgraded: every client\.evaluate\(\) raises AttributeError; changelog has no migration entry \(${source}\)\. Final state: file an issue or pin v1\.$`),
];
if (observed.length !== observation.length || observed.some((line, i) => !observation[i]!.test(line))) return false;
// Consume the complete decision explanation too. Additional work under a
// valid heading or in a choice description must remain substantive.
const range = `D${first}–D${last}`;
const tail = parts[4]!.replace(new RegExp(`D${first}[–-]D${last}`, 'g'), range).split('\n');
const expected = [
'ELI10: Each numbered point is a place a real first-time user stops and asks a question nobody is there to answer. The plan should remove every one it reasonably can.',
'Stakes if we pick wrong: leave one in and that is the step where the developer\'s session ends; each maps to a contract PLAN.md explicitly asked to be reviewed.',
`Recommendation: A because all five map one-to-one to the ${range} decisions you already resolved as "fix in plan", so addressing all of them is consistent with those calls.`,
'Completeness: A=10/10, B=depends on selection, C=6/10, D=1/10',
'A) All of them, fix every confusion point (recommended)',
`✅ Consistent with ${range}; every confusion point already has an agreed fix`,
'✅ Leaves no known dead end in the first 30 minutes of use',
'❌ Full set of fixes touches README, client.py, demo gate, error class, and changelog (human: ~3 days / CC: ~1.5 hours)',
'B) Let me pick which ones matter',
'✅ Lets you drop a point if you know something the docs do not show',
'✅ Keeps the plan focused on what you consider blocking',
`❌ Reopens decisions ${range} that were just settled`,
'C) The critical ones only (#1, #2, #5), skip #3 and #4',
'✅ Covers the broken quickstart, the TTHW blocker, and the upgrade break',
'✅ Smaller diff to review',
'❌ Ships an inconsistent API and an undiagnosable auth error in a DX polish release',
'D) This is unrealistic, our developers already know the context',
'✅ Zero work now',
'✅ Valid if every beta user is internal and already trained',
'❌ README.md:3-5 describes an external ML engineer meeting the SDK fresh, which contradicts this',
'Net: trading a known, already-scoped set of fixes against leaving a documented dead end in the first session.',
];
return tail.length === expected.length && tail.every((line, i) => line.trim() === expected[i]);
}
/** A batched native call remains one decision; the caller owns call-ID deduplication. */
export function isDevexReviewIssue(fp: AskUserQuestionFingerprint, priorCalls: readonly NativePlanQuestionCall[] = []): boolean {
if (answeredQuotedAccuracy(fp) || answeredRoleplayRecap(fp, priorCalls)) return false;
return answeredMeasurementGate(fp) || answeredSetupRepair(fp) || answeredContractRepair(fp) || answeredKeylessDemoRepair(fp) || answeredDocumentationFollowup(fp) || questionRecords(fp, true).some(substantiveIssue);
}
/** Key acquisition docs and eliminating the demo's key requirement are distinct work. */
function answeredKeylessDemoRepair(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (call?.answered !== true || call.failed !== false || call.questions.length !== 1 ||
!Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length ||
fp.signature !== `${call.sessionId}:${call.toolUseId}`) return false;
const q = call.questions[0]!;
if (q.multiSelect || q.options.length < 2 || new Set(q.options.map(o => o.label)).size !== q.options.length ||
fp.options.length !== q.options.length || fp.options.some((o, i) => o.index !== i + 1 || o.label !== q.options[i]!.label) ||
!/^Golden path$/i.test(q.header.trim()) || /<gstack-qid/i.test(q.question)) return false;
const selected = q.options.filter(o => o.label === call.answers?.[q.question]);
if (selected.length !== 1 || !/^Install, demo, then key(?: \(recommended\))?$/i.test(selected[0]!.label) ||
!/\bDemo path is guaranteed keyless and offline; if the runtime currently insists on a key for the demo, remove that check\b/.test(selected[0]!.description ?? '')) return false;
const lines = q.question.split('\n');
return /^D\s*\d+\s*[—–:-]\s*Pass 1 Getting Started \((?:10|[0-9])\/10 today\): should the golden path put the demo BEFORE the API key step\?$/i.test(lines[0]!) &&
/^ELI10: Today README "Getting started" \(lines \d+-\d+\) reads install, set [A-Z][A-Z_]+, run a missing file\./m.test(q.question);
}
/** New documentation and example obligations are separate from the original repairs. */
function answeredDocumentationFollowup(fp: AskUserQuestionFingerprint): boolean {
const call = fp.nativeCall;
if (!call?.answered || call.failed !== false || call.questions.length !== 1 ||
!Array.isArray(call.unansweredQuestionIndices) || call.unansweredQuestionIndices.length ||
fp.signature !== `${call.sessionId}:${call.toolUseId}`) return false;
const q = call.questions[0]!;
if (q.multiSelect || q.options.length < 2 || new Set(q.options.map(o => o.label)).size !== q.options.length ||
administrativeQuestion(q.header, q.question, q.options)) return false;
const selected = q.options.filter(o => o.label === call.answers?.[q.question]);
const ids = [...q.question.matchAll(/<gstack-qid:([^>]+)>/gi)];
if (selected.length !== 1 || ids.length !== 1 || (q.question.match(/<gstack-qid/gi)?.length ?? 0) !== 1) return false;
const label = selected[0]!.label.trim().replace(/^[A-Z][.):]\s*/i, '').replace(/\s*\(recommended\)$/i, '');
const headline = q.question.split('\n')[0]!.replace(/<gstack-qid:[^>]+>/i, '').trim();
if (/^plan-devex-review-todo\d+-migration-guide$/i.test(ids[0]![1]!) && /^TODO[- ]\d+ Migration$/i.test(q.header.trim())) {
// A written upgrade guide is additional work beyond the accepted runtime
// compatibility shim. Require that distinct gap and the selected doc task;
// a recap, hypothetical example or unselected guide cannot supply it.
const parts = q.question.replace(/\s*<gstack-qid:[^>]+>\s*$/i, '').trim().split(/\n\s*\n/);
const compact = (text: string | undefined) => (text ?? '').replace(/\s+/g, ' ').trim();
return call.answered === true && parts.length === 5 &&
/^D\s*\d+\s*[—–:-]\s*TODO: should the plan include a v\d+→v\d+ written migration guide\?$/i.test(compact(parts[0])) &&
/^The deprecation shim \(T\d+\) handles the runtime experience: v\d+ callers get a DeprecationWarning naming `[a-z_]\w*\(\)` as the replacement\. But there is currently no written migration guide in docs\/\.$/i.test(compact(parts[1])) &&
/^A one-page migration guide covers: - What changed \(`[a-z_]\w*\(\)` → `[a-z_]\w*\(\)`\) - What stayed the same \(all other APIs\) - How to find and update callsites \(grep for `[^`\n]+`\) - When the shim is removed \(e\.g\., v\d+(?:\.\d+)?\)$/i.test(compact(parts[2])) &&
/^Without it, developers upgrading a large codebase need to discover the change at each call site rather than planning the migration upfront\. The changelog has the what; the guide provides the how and the timeline\.$/i.test(compact(parts[3])) &&
/^Completeness: A=(?:10|[0-9])\/10 \(complete\), B=(?:10|[0-9])\/10 \(runtime-only, no planning\), C=(?:10|[0-9])\/10$/i.test(compact(parts[4])) &&
/^Add to TODOS\.md [—–-] include migration guide in plan$/i.test(label) &&
/^Add docs\/migration-v\d+-v\d+\.md as a P[0-3] task\. One page covering the rename, unchanged APIs, grep command to find callsites, and shim removal timeline\. Completeness: (?:10|[0-9])\/10\.$/i.test(compact(selected[0]!.description));
}
if (ids[0]![1] === 'devex-api-key-docs' && /^API key docs$/i.test(q.header.trim())) {
// The earlier auth-error decision changes runtime diagnostics. This one
// adds the missing acquisition instructions to the README itself.
return /^D\s*\d+\s*[—–:-]\s*Pass\s+\d+:\s*Documentation\s*[—–:-]\s*README says ['"][^'"]+['"] but never says where to get one\.$/i.test(headline) &&
/^Add key acquisition link to README$/i.test(label);
}
if (ids[0]![1] === 'devex-todo-real-world-examples' && /^TODO examples$/i.test(q.header.trim())) {
// The quickstart repair supplies one missing file. These additional
// custom-data examples are an independently accepted follow-up obligation.
return /^D\s*\d+\s*[—–:-]\s*TODO check:\s*Real-world examples beyond the bundled sample data\?$/i.test(headline) &&
/^\*\*What:\*\* Add \d+(?:-\d+)? additional examples\/ files showing real use cases\b/m.test(q.question) &&
/^(?:Add to TODOS\.md for post-beta|Build it now as part of this plan)$/i.test(label);
}
return false;
}
/** Select POLISH only on the recognized mode menu; leave all other answers unchanged. */
export function devexReviewModePick(fp: AskUserQuestionFingerprint): number | null {
if (fp.nativeCall && fp.nativeCall.questions.length !== 1) return null;
const record = questionRecords(fp)[0];
const text = record ? `${record.header} ${record.question}` : '';
if (!/<gstack-qid:plan-devex-review-mode>/i.test(text)
&& !/how\s*deep\s*should\s*this\s*dx\s*review|which\s*(?:dx\s*)?review\s*mode/i.test(text)) return null;
const modes = fp.options.map(option => ({
index: option.index,
mode: /^(?:[A-C][.)])?DX(POLISH|EXPANSION|TRIAGE)(?:$|[^A-Z])/.exec(
option.label.split(/[│┌\r\n]/, 1)[0]!.replace(/\s+/g, '').toUpperCase(),
)?.[1],
}));
if (!['POLISH', 'EXPANSION', 'TRIAGE'].every(mode => modes.filter(option => option.mode === mode).length === 1)) return null;
return modes.find(option => option.mode === 'POLISH')!.index;
}
-438
View File
@@ -1,438 +0,0 @@
import type { NativePlanQuestion, PlanCountTranscript } from './plan-count-transcript';
export const DEVEX_SEEDED_GAPS = [
'local-ci-gate', 'missing-quickstart', 'reversed-arguments', 'opaque-auth-error', 'breaking-upgrade',
] as const;
export type DevexSeededGap = typeof DEVEX_SEEDED_GAPS[number];
/** Bind an unnamed signature question to its own first asserted explanation. */
function explainedReversedSignatures(q: NativePlanQuestion, title: string): boolean {
const question = /^(?:Journey stage [A-Z ]+: )?the two public functions take the same two arguments in (?:opposite|reversed) positional order\. How should (?:the plan|we) (?:fix|align|unify) the signatures\?$/i.test(title);
const traced = /^Journey stage: REAL USAGE\. Two sibling functions take the same two arguments in (?:opposite|reversed) order\.$/i.test(title);
const declared = traced || /^Journey stage(?: REAL USAGE:|: REAL USAGE\.) The two public evaluation functions take the same two arguments in (?:opposite|reversed) order\.$/i.test(title);
// A dedicated assertion can put its named signatures in its own Evidence
// field. Bind subject, source identities and repair instead of menu wording.
const subject = title.replace(/^Journey stage(?: REAL USAGE:|: REAL USAGE\.)\s*/i, '');
const evidenced = !question && !declared &&
/^(?:the )?(?:two|both) (?:public (?:evaluation )?|evaluation )functions take\b/i.test(subject) &&
/\bthe same two arguments\b/i.test(subject) && /\b(?:opposite|reversed) (?:positional )?order\.?$/i.test(subject);
const declaration = declared || evidenced;
if (!question && !declaration) return false;
const lines = q.question.split('\n');
if (lines[0]!.trim().replace(/^D\s*\d+\s*[—–:-]\s*/i, '') !== title) return false;
const explanation = lines.findIndex(line => line.startsWith('ELI10: '));
const context = lines.slice(1, explanation).filter(line => line.trim());
const project = traced ? /^Project\/branch\/task: [^;\n]+; ([\w./-]+):\d+(?:[-–]\d+)?\.$/.exec(context[0] ?? '') : declaration && context.length === 1
? /^Project\/branch\/task: [^;\n]+; ([\w./-]+) lines? \d+(?: to |[-–])\d+\.$/.exec(context[0]!) : null;
if (explanation < 1 || (declared && !project) || (!traced && !evidenced && context.some(line =>
(!declaration && !/^Project\/branch\/task: [^;\n]+; reviewing the public function signatures in [\w./-]+\.$/.test(line)) ||
/\b(?:quoted|source excerpt|source example|hypothetical|historical|not (?:a )?current|if approved)\b/i.test(line)))) return false;
// Inline code may name each signature; a quoted/fenced explanation, earlier
// unrelated sentence, past definition or hypothetical definition cannot.
const declaredSignatures = traced
? /^I traced the first real integration after the demo\. ([\w./-]+) lists the two evaluation functions: (`?)run_eval\(\s*dataset\s*,\s*evaluator\s*\)\2 and (`?)run_batch\(\s*evaluator\s*,\s*dataset\s*\)\3\./.exec(context[1] ?? '')
: declaration && /^ELI10: ([\w./-]+) documents (`?)run_eval\(\s*dataset\s*,\s*evaluator\s*\)\2 and (`?)run_batch\(\s*evaluator\s*,\s*dataset\s*\)\3\. Same two concepts, reversed positional order, and neither function requires keywords\./.exec(lines[explanation]!);
if (!evidenced && (declaration ? !declaredSignatures || declaredSignatures[1] !== project?.[1]
: !/^ELI10: [\w./-]+(?: lines? \d+(?:\s*[-–]\s*\d+)?)? define (`?)run_eval\(\s*dataset\s*,\s*evaluator\s*\)\1 and (`?)run_batch\(\s*evaluator\s*,\s*dataset\s*\)\2\./.test(lines[explanation]!))) return false;
const currentProse = (text: string) => {
let fence = false;
return text.split('\n').filter(line => {
if (/^\s*(?:```|~~~)/.test(line)) { fence = !fence; return false; }
return !fence && !/^\s*>/.test(line);
}).join('\n')
.replace(/(^|[.!?\n]\s*)((?:Correction:\s*)?(?:this|that|the) (?:evidence|trace) (?:is|was|has been) )["“'‘`](withdrawn|rejected|cancelled|canceled|superseded|historical|hypothetical|(?:not|no longer) current)["”'’`]/gi, '$1$2$3')
.replace(/`[^`\n]*`|"[^"\n]*"|“[^”\n]*”/g, '');
};
// The traced declaration owns its named signatures before ELI10, so its
// currentness must include that same source paragraph.
const current = currentProse(lines.slice(traced || evidenced ? 1 : explanation).join('\n'));
if ((current.match(/^ELI10:/gm)?.length ?? 0) !== 1) return false;
if (declaration && /\b(?:if|once|when|unless) (?:approved|accepted)|\b(?:after|pending) approval\b/i.test(current)) return false;
if (declaration && /(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:these|the) (?:functions|signatures) (?:are (?:now|already)|have been) (?:aligned|consistent)\b/i.test(current)) return false;
if ((traced || evidenced) && /(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:this|that|the) (?:trace|evidence) (?:is|was|has been) (?:withdrawn|rejected|cancelled|canceled|superseded|historical|hypothetical|(?:not|no longer) current)\b/i.test(current)) return false;
if (/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:(?:this|that|the) (?:finding|explanation)|(?:(?:this|that|the) )?argument[- ]order (?:issue|defect)|these signatures)\b[^.\n]*\b(?:withdrawn|rejected|(?:already )?(?:fixed|resolved)|historical|(?:not|no longer) current)\b/i.test(current) ||
/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:there is|there's) no argument[- ]order (?:issue|defect)\b/i.test(current) ||
/(?:^|[.!?\n]\s*)(?:Correction:\s*)?run_eval and run_batch now (?:use|take) the same positional order\b/i.test(current)) return false;
if (evidenced) {
// Only an asserted citation at the start of this decision's field owns
// the pair; quoted examples, later borrowed prose and split fields do not.
const fields = lines.slice(1, explanation + 1).filter(line => /^(?:Evidence|ELI10):/.test(line));
const pair = /^(?:Evidence|ELI10):\s*[\w./-]+(?: lines? \d+(?:\s*(?:[-–]|to)\s*\d+)?|:\d+(?:[-–]\d+)?)?:\s*(`?)run_eval\(\s*dataset\s*,\s*evaluator\s*\)\1 and (`?)run_batch\(\s*evaluator\s*,\s*dataset\s*\)\2(?:[.;]|$)/;
if (!fields.some(line => pair.test(line)) || fields.some(line =>
/^(?:Evidence|ELI10):\s*(?:>|`|"|“|Source\b|Quoted\b|Historical\b|Earlier\b|Example\b|Hypothetical\b|If\b|Assuming\b|Provided\b)/i.test(line))) return false;
return q.options.some(option => {
const label = currentProse(option.label.replace(/`(\(\s*dataset\s*,\s*evaluator\s*\))`/g, '$1'));
const remedy = currentProse(option.description ?? '');
const first = remedy.split(/[.!?\n]/)[0] ?? '';
// The named pair above owns "Both" and the run_x signature shorthand.
// This offered repair enforces the same keyword-only shape on that pair,
// retaining a warning for existing positional callers during the beta.
const stagedKeywords = /^(?:[A-D]\)\s*)?(?:Align|Unify|Standardize) order \+ keyword-only with beta deprecation(?: \(recommended\))?$/i.test(label) &&
/^Both(?: functions)? become run_x\(\*\s*,\s*dataset\s*,\s*evaluator\s*\)\. Positional (?:calls )?accepted for one beta cycle with a DeprecationWarning naming the fix\.$/i.test(remedy);
return (stagedKeywords || (/^(?:[A-D]\)\s*)?(?:Align|Unify|Standardize)\b/i.test(label) && /\(\s*dataset\s*,\s*evaluator\s*\)/.test(label) &&
/\bsame (?:positional )?order\b/i.test(first) && /\bboth functions\b/i.test(first) &&
/\bkeywords? (?:accepted|supported)\b|\baccept keywords\b/i.test(remedy) &&
/\bswaps? (?:is |are )?(?:detected|caught|rejected)\b/i.test(remedy) && /\b(?:clear|actionable) (?:error|message)\b/i.test(remedy))) &&
!/\b(?:if|unless|when|once|after|pending)\b|\b(?:no|not|never|without|do not|don't)\b|\b(?:other|another|foreign|different) (?:functions?|API|pair|project|issue)\b/i.test(`${label}\n${remedy}`) &&
!/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:this|that|the) (?:option|action|correction) (?:is|was|has been) (?:historical|withdrawn|rejected|cancelled|canceled|superseded|(?:not|no longer) current)\b/i.test(remedy) &&
(stagedKeywords || !/\brun_(?!eval\b|batch\b)\w+\b/.test(remedy));
});
}
// A declared reversal may offer a keyword-only repair instead of a swap
// guard. It must bind both arguments to both functions in the same option.
if (declaration) return q.options.some(option =>
(traced ? /^Fix in plan: same order \+ keyword-only for both(?: \(recommended\))?$/i.test(option.label) &&
/^✅\s*run_eval\(\*\s*,\s*dataset\s*,\s*evaluator\s*\) and run_batch\(\*\s*,\s*dataset\s*,\s*evaluator\s*\); wrong order becomes a TypeError naming the parameter at the call site\b/i.test(option.description ?? '')
: /^(?:Align|Unify|Standardize) order \+ keyword-only(?: \(recommended\))?$/i.test(option.label) &&
/^Both functions (?:take|accept|use) dataset and evaluator as keyword-only in the same order\./i.test(option.description ?? '')) &&
!/\b(?:if|once|when|unless) (?:approved|accepted)|\b(?:after|pending) approval\b/i.test(currentProse(option.description ?? '')) &&
!/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:(?:do not|don't|never) (?:change|align|unify) (?:either|both|the|these) (?:functions?|signatures?)\b|(?:do not|don't|never) (?:make|require) (?:either|both|the) (?:functions?|signatures?|arguments?) keyword-only\b|(?:this|the) (?:option|correction|action) is (?:withdrawn|rejected|cancelled)\b)/i.test(currentProse(option.description ?? '')));
// The same offered action must align both functions and retain the call-site
// swap guard. Selecting an offered alternate or deferral is still a decision.
return q.options.some(option => /^Same order\s*\+\s*swap guard(?: \(recommended\))?$/i.test(option.label) &&
/^(?:✅\s*)?Both (?:become|use|take) `?\(\s*dataset\s*,\s*evaluator\s*\)`?, accept keywords, and raise a call-site `?TypeError`? naming the swapped argument and the fix if types are reversed\./i.test(option.description ?? '') &&
!/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:(?:do not|don't|never) (?:change|align|unify) (?:either|both|the) signatures?\b|(?:do not|don't|never|skip) (?:add|require|implement) (?:a |the )?swap guard\b|(?:this|the) (?:option|correction|action) is (?:withdrawn|rejected|cancelled)\b)/i.test(currentProse(option.description ?? '')));
}
/** Identify a dedicated seed decision by its subject and meaningful alternatives. */
function decisionGaps(q: NativePlanQuestion): DevexSeededGap[] {
const rawTitle = q.question.split('\n')[0]!.trim().replace(/^D\s*\d+\s*[—–:-]\s*/i, '');
const title = rawTitle.replace(/`([^`\n]+)`/g, '$1');
// Recording later work after the current repair is settled is a backlog
// disposition, not the required decision about the current seeded gap.
if (/^(?:TODO|Follow[- ]up)\s*[:—–-]/i.test(title) &&
/\b(?:later|future) release\b|\bbacklog\b/i.test(title)) return [];
const questionMarks = title.match(/\?/g)?.length ?? 0;
const upgradeVocabulary = /\b(?:alias|warning|compatibility|deprecat\w*|migration|remov\w*|rename|keep)\b/i.test(title);
// A named method becoming its replacement is a transition even when the
// title asks about a soft landing. Its own explanation must establish the gap.
const upgradeTransition = !upgradeVocabulary && /\bClient\.evaluate\(\) becomes Client\.run\(\)/i.test(title);
// A question may name the journey problem and put its asserted source facts
// in ELI10. Topic words alone never supply either defect or its remedy.
const explainedAuthentication = /^(?:(?:Authentication (?:error|failure)|API[- ]key (?:error|failure|rejection)):\s*)?(?:what|how)\b/i.test(title) &&
/\b(?:API[- ]key|authentication)\b/i.test(title) && /\b(?:rejected|invalid|error|failure)\b/i.test(title);
const explainedUpgrade = !upgradeVocabulary && !upgradeTransition && /\bClient\.evaluate\(\)/.test(title) &&
(/\bupgrade\b/i.test(title) || /Client\.run\(\)/.test(title));
const explainedSubject = explainedAuthentication || explainedUpgrade;
// Journey labels, possessives and a positive inclusive aside format the
// asserted subject. Keep the original title for all meaning/currentness checks.
// These six stages come from the skill's journey trace. A decision may span
// adjacent touchpoints without changing the subject or who asserts it.
const journeyStage = '(?:DISCOVER|INSTALL|HELLO WORLD|REAL USAGE|DEBUG|UPGRADE)';
const stage = title.match(new RegExp(`^Journey stage ${journeyStage}(?:\\s*\\/\\s*${journeyStage})?: (.+)$`, 'i'))
?? title.match(new RegExp(`^Journey stage: ${journeyStage}(?:\\s*\\/\\s*${journeyStage})?\\. (.+)$`, 'i'));
// New field declarations require a canonical stage. Existing direct
// questions can still name another touchpoint without normalizing it.
if (!stage && (/^Journey stage:/i.test(title) ||
(/^Journey stage\b/i.test(title) && !title.includes('?')))) return [];
if (stage && /^(?:Assuming|Provided)\b/i.test(stage[1]!.trim())) return [];
const assertionTitle = stage ? stage[1]!
.replace(/\b([A-Za-z0-9_.]+)['’]s\b/g, '$1')
.replace(/, including ([A-Za-z0-9_-]+(?: [A-Za-z0-9_-]+){0,6}),/gi, (aside, subject: string) =>
/\b(?:if|unless|assuming|provided|except|excluding|only|no|not|never|without|was|were|is|are|has|had|may|might|could|would|historical|earlier|quoted|source|example|hypothetical|fixed|resolved|cancelled|canceled|withdrawn|rejected|superseded)\b/i.test(subject) ? aside : '')
: title;
// Journey cards may state the defect in the title and place its named
// current contract in Evidence/ELI10. Keep this route owned even on rejection.
const evidenceJourney = Boolean(stage && /^Evidence:/m.test(q.question)) &&
(/\bquickstart\b/i.test(assertionTitle) ? 'missing-quickstart'
: /\bCI (?:check|gate)\b/i.test(assertionTitle) ? 'local-ci-gate' : undefined);
const opaqueAuthentication = /^(?:The )?authentication error says nothing[.?]?$/i.test(assertionTitle);
const vanishingUpgrade = /^v\d+ Client\.evaluate\(\) vanishes in v\d+ with no warning, alias, or guide[.?]?$/i.test(assertionTitle);
// A defect heading can assert a prerequisite or compare named signatures
// without a finite verb. Keep these semantic families narrow: a topic label,
// healthy signature pair or optional check is not an asserted defect.
const nominalDefect = /^(?:The )?(?:Mandatory|Required) (?:\d+(?:\.\d+)?[- ](?:minute|second) )?(?:remote )?CI (?:check|gate) (?:before|gates) (?:the )?first local (?:result|evaluation|run)[.?]?$/i.test(assertionTitle) ||
/^run_eval\(\s*dataset\s*,\s*evaluator\s*\) (?:vs\.?|versus|and) run_batch\(\s*evaluator\s*,\s*dataset\s*\): (?:reversed|opposite|swapped) (?:positional|argument) order[.?]?$/i.test(assertionTitle);
const nominalSubject = /^(?:Mandatory|Required|Optional)\b[^?!\n]*\bCI (?:check|gate)\b/i.test(assertionTitle) ||
/^(?:The )?(?:Mandatory|Required|Optional)\b[^?!\n]*\bCI (?:check|gate) gates\b/i.test(assertionTitle) ||
/^run_eval\([^)]+\) (?:vs\.?|versus|and) run_batch\([^)]+\):/i.test(assertionTitle);
if (nominalSubject && !nominalDefect && !evidenceJourney) return [];
const signatureDeclaration = /^run_eval\(\s*dataset\s*,\s*evaluator\s*\) and run_batch\(\s*evaluator\s*,\s*dataset\s*\) (?:take|takes)\b/i.test(assertionTitle);
if (/^run_eval\([^)]+\) and run_batch\([^)]+\) (?:take|takes)\b/i.test(assertionTitle) && !signatureDeclaration) return [];
// The named tuples can establish the reversal without an adjective. Keep
// their identities and order together; malformed or negated comparisons
// cannot fall through to the broader direct-question path.
const tupleSubject = /^run_eval (?:takes?|does not take)\b[^\n]*\brun_batch\b/i.test(assertionTitle);
const tuples = /^run_eval takes\s*\(\s*(\w+)\s*,\s*(\w+)\s*\) (?:but|while) run_batch takes\s*\(\s*(\w+)\s*,\s*(\w+)\s*\)(?:[.?]|\. Fix in plan\?)?$/i.exec(assertionTitle);
const reversedTuples = Boolean(tuples && tuples[1] !== tuples[2] &&
[tuples[1], tuples[2]].sort().join(',') === 'dataset,evaluator' &&
tuples[1] === tuples[4] && tuples[2] === tuples[3]);
if (tupleSubject && !reversedTuples) return [];
const finiteTitle = signatureDeclaration ? assertionTitle.replace(/\([^)]*\)/g, '') : assertionTitle;
// Negative availability asserts a missing referenced file. Bind it to that
// object; do not erase a negation of the quickstart's own reference or gate.
const absentReference = /\b(?:points?|references?) (?:at|to) (?:examples\/first_eval\.py|(?:a|the) (?:file|example)),? (?:which|that) (?:is not (?:shipped|in (?:the )?(?:package|wheel)(?: or (?:the )?(?:release )?examples archive)?)|does not (?:ship|exist))[.?]?$/i.test(assertionTitle);
const newAssertion = nominalDefect || signatureDeclaration || reversedTuples || absentReference || opaqueAuthentication || vanishingUpgrade;
const guardedDeclaration = Boolean(stage || newAssertion || upgradeTransition || explainedSubject);
const polarityTitle = absentReference ? title.replace(/\bdoes not (ship|exist)([.?]?)$/i, 'is absent$2') : title;
// Punctuation cannot route a newly admitted asserted family around its
// ownership checks; an offered alternate still resolves the same decision.
const declaration = (questionMarks === 0 || (questionMarks === 1 && title.endsWith('?'))) &&
(/^(?:[A-Za-z0-9_.]+\s+){1,12}(?:points?|references?|blocks?|requires?|takes?|raises?|removes?|drops?)\b/i.test(finiteTitle) || nominalDefect || opaqueAuthentication || vanishingUpgrade) &&
!/^`[^`]*`$/.test(rawTitle) &&
!/\b(?:if|unless|suppose|might|may|could|would|previously|earlier|historical|hypothetical|example|quoted|source|never|no longer|does not|do not|did not)\b/i.test(polarityTitle);
if (newAssertion && !declaration && !evidenceJourney) return [];
if ((!declaration && (!title.endsWith('?') || questionMarks !== 1)) ||
/^`[^`]*`[.?]?$/.test(rawTitle) ||
/^(?:>|"|“|Example\b|Quoted\b|Source(?: excerpt| example)?[,:.]|Historical\b|Earlier review\b|If (?:approved|accepted)\b|Assuming\b|Provided\b|Suppose\b)|\bhypothetical\b/i.test(title) ||
/\b(?:if|once|when|unless) (?:approved|accepted)|\b(?:after|pending) approval\b/i.test(title) ||
/\b(?:continue|proceed|next section|move on|format|already (?:fixed|resolved))\b/i.test(title) ||
/\b(?:have|did)\b[^?]*\bread\b|\b(?:narrative|trace|recap|summary)\b[^?]*\b(?:accurate|match|confirm)\b/i.test(title) ||
/\b(?:report|summary|recap)\b[^?]*\b(?:mention|include|reference|list)\b|\b(?:mention|include|reference|list)\b[^?]*\b(?:report|summary|recap)\b/i.test(title)) return [];
let offered = q.options;
let signatureOptions = q.options;
if (declaration || upgradeTransition || explainedSubject || evidenceJourney) {
const currentProse = (text: string, offeredAction = false) => {
// In a tuple decision, a semicolon also separates current assertions.
// Quotations and fenced examples are still removed as whole statements.
if (reversedTuples || upgradeTransition) text = text.replace(/;/g, '.');
// An option's trailing effort estimate separates its prose from an owned
// status even without punctuation. Keep it on the same line so a quoted
// historical sentence is still removed as one quotation below.
const bounded = guardedDeclaration && offeredAction ? text.replace(/(\(human:[^()\n]{1,80}\/ CC:[^()\n]{1,80}\))[ \t]+(?=(?:Correction:\s*)?(?:(?:this|that|the) (?:option|action|correction)|D\s*[1-9]\d*) (?:is|was|has been) ["“'‘`]?(?:cancelled|canceled|superseded|withdrawn|rejected|(?:not|no longer) current)\b)/gi, '$1. ') : text;
// Preserve a scalar status asserted by a current, unquoted owner before
// removing source quotations. The owner must still match this decision.
const owned = guardedDeclaration ? bounded.replace(/(^|[.!?\n]\s*)((?:Correction:\s*)?(?:(?:this|that|the) (?:finding|issue|gap|defect|explanation|option|action|correction)|D\s*[1-9]\d*) (?:is|was|has been) )["“'‘`](cancelled|canceled|superseded|withdrawn|rejected|(?:not|no longer) current)["”'’`]/gim, '$1$2$3') : bounded;
let fence = false;
return owned.split('\n').filter(line => {
if (/^\s*(?:```|~~~)/.test(line)) { fence = !fence; return false; }
return !fence && !/^\s*>/.test(line);
}).join('\n')
.replace(/(^|[.!?\n]\s*)((?:Correction:\s*)?(?:this|that|the) (?:finding|issue|gap|defect|explanation|option|action|correction) (?:is|was|has been) )["“](withdrawn|rejected|(?:already )?(?:fixed|resolved)|historical|(?:not|no longer) current|cancelled)["”]/gim, '$1$2$3')
.replace(/`[^`\n]*`|"[^"\n]*"|“[^”\n]*”/g, '');
};
const sourceFrame = /(?:^|[.!?\n;]\s*)(?:(?:ELI10|Project\/branch\/task):\s*)?(?:(?:Source(?: excerpt| example)?|Quoted(?: source| example)?|Historical(?: example| assessment)?(?: only)?|Earlier(?: review)? assessment|Example|Hypothetical(?: example| assessment| scenario)?|If approved|If accepted)[,:.]|(?:The following|This assessment|This explanation)\b[^.\n]*\b(?:quoted|source|historical|hypothetical|example)\b|Historically,)/i;
const lines = q.question.split('\n'), explanation = lines.findIndex(line => /^ELI10:/.test(line));
const preface = lines.slice(0, explanation < 0 ? undefined : explanation + 1).join('\n');
if (guardedDeclaration && /^(?:Project\/branch\/task|ELI10):\s*(?:Assuming|Provided)\b/im.test(currentProse(preface))) return [];
if (/^\s*(?:```|~~~)/m.test(preface) || sourceFrame.test(currentProse(preface)) ||
/\bnot (?:a )?current (?:finding|issue|defect)\b/i.test(currentProse(preface))) return [];
const current = currentProse(q.question);
const approval = /\b(?:if|once|when|unless) (?:approved|accepted)|\b(?:after|pending) approval\b/i;
if (explainedSubject) {
const project = lines.filter(line => /^Project\/branch\/task:/.test(line));
const raw = (lines[explanation] ?? '').replace(/^ELI10:\s*/, '');
if (explanation < 1 || lines.filter(line => /^ELI10:/.test(line)).length !== 1 ||
project.length !== 1 || !/^Project\/branch\/task:\s*EvalKit\b/i.test(project[0]!) ||
/\b(?:other|another|foreign|different) (?:project|SDK|codebase|repository)\b/i.test(project[0]!) ||
/^[>"“'‘\x60]|^(?:Source|Quoted|Historical|Earlier|Example|Hypothetical|Assuming|Provided)\b/i.test(raw) ||
approval.test(current)) return [];
// Preserve the exact error payload as data, while whole quoted/fenced
// explanations are still removed by the existing prose guard.
const marker = 'GSTACK_OWNED_AUTH_LITERAL';
if (raw.includes(marker)) return [];
const literal = /(?<![\w.])(\x60?)AuthError\((["'])request failed\2\)\1/g;
const literalCount = [...raw.matchAll(literal)].length;
const asserted = currentProse(raw.replace(literal, marker)
.replace(/\x60((?:Client\.|client\.)?(?:evaluate|run)\((?:\.\.\.)?\))\x60/g, '$1'));
const sourceStatus = currentProse(q.question.replace(/(^|[.!?\n]\s*)((?:Correction:\s*)?(?:this|that|the) (?:statement|evidence|source claim) (?:is|was|has been) )["“'‘\x60](withdrawn|rejected|cancelled|canceled|superseded|historical|hypothetical|(?:not|no longer) current)["”'’\x60]/gi, '$1$2$3'));
if (/(?:^|[.!?\n]\s*)(?:Previously|Earlier|Historically)\b/i.test(asserted) ||
/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:this|that|the) (?:statement|evidence|source claim) (?:is|was|has been) (?:withdrawn|rejected|cancelled|canceled|superseded|historical|hypothetical|(?:not|no longer) current)\b/i.test(sourceStatus)) return [];
const sentences = asserted.split(/(?<=[.!?])\s+/);
const ownsSource = (fact: string) => {
const citations = (project[0]! + ' ' + fact).match(/[\w./-]+\.md:\d+(?:[-–]\d+)?/g) ?? [];
return citations.length > 0 && citations.every(citation => citation.startsWith('docs/api.md:'));
};
if (explainedAuthentication) {
const index = sentences.findIndex(sentence => /GSTACK_OWNED_AUTH_LITERAL/.test(sentence));
const fact = sentences[index] ?? '', preceding = sentences[index - 1] ?? '';
const event = /^(?:(?:If|When) (.+),\s*)?(?:(?:the )?SDK (?:raises|throws)|they (?:get|receive)) GSTACK_OWNED_AUTH_LITERAL(?:\s*\([\w./-]+\.md:\d+(?:[-–]\d+)?\))?\.?$/i.exec(fact);
const condition = event?.[1]?.replace(/\bnot exported\b/gi, 'missing');
const states = condition?.replace(/^(?:that|the|an? API) key is\s+/i, '');
const keyState = states && states !== condition &&
/\b(?:stale|mistyped|revoked|rejected|missing|invalid)\b/i.test(states) &&
states.replace(/\b(?:stale|mistyped|revoked|rejected|missing|invalid|or|and|simply)\b|[\s,]/gi, '') === '';
const pastedKey = condition && /^they (?:paste|enter) it wrong(?:,? or it was revoked)?$/i.test(condition) &&
/^(?:The first thing a developer does after the demo is|The developer) (?:paste|pastes|enter|enters) a key\.$/i.test(preceding);
const ambiguity = sentences[index + 1] ?? '';
if (literalCount !== 1 || !event || !ownsSource(fact) ||
(condition && !keyState && !pastedKey) ||
!/^(?:['‘]?request failed['’]?|(?:this|the) (?:error|message)|That) could mean\b[^.?!]*\b(?:DNS|proxy|rate limit|key|network|server)\b/i.test(ambiguity)) return [];
} else {
const [baseline = '', transition = ''] = sentences;
const namedTransition = /^(?:Version 1|v1) (?:exposes|provides) Client\.evaluate\(\)\.$/i.test(baseline) &&
/^(?:2\.0|v2) renames (?:it|Client\.evaluate\(\)) to Client\.run\(\) and (?:deletes|removes|drops) (?:the )?old (?:name|method)\b/i.test(transition) &&
/\bno alias\b/i.test(transition) && /\bno warning\b/i.test(transition) && /\bno migration guide\b/i.test(transition);
const runtimeBreak = /^(?:Your persona|The developer) (?:wires EvalKit into|uses EvalKit in) production CI\.$/i.test(baseline) &&
/^When they (?:bump|upgrade) to (?:2\.0|v2), every client\.evaluate\(\.\.\.\) call (?:dies|fails) with a generic AttributeError that names nothing about run\(\)\./i.test(transition);
if ((!namedTransition && !runtimeBreak) || !ownsSource(transition) ||
/\b(?:if|unless|assuming|provided|might|may|could|would|historical|hypothetical|not|never|no longer)\b/i.test(transition)) return [];
}
}
if (upgradeTransition) {
if (/^ELI10:\s*>/.test(lines[explanation] ?? '')) return [];
const namedCurrent = currentProse(q.question.replace(/`([A-Za-z_$][\w.$]*(?:\(\))?)`/g, '$1'));
if (/(?:^|[.!?\n]\s*)(?:Correction:\s*)?Client\.evaluate\(\) (?:is (?:now|already|still)|now remains) (?:a |an )?(?:deprecated |compatibility )?alias\b/i.test(namedCurrent)) return [];
const first = currentProse((lines[explanation] ?? '').replace(/`([A-Za-z_$][\w.$]*(?:\(\))?)`/g, '$1'))
.replace(/^ELI10:\s*/, '').split(/(?<=[.!?])\s/)[0] ?? '';
if (explanation < 1 || lines.filter(line => /^ELI10:/.test(line)).length !== 1 || approval.test(current) ||
/\b(?:if|unless|assuming|provided|suppose|might|may|could|would|previously|earlier|historical|hypothetical|never|no longer|does not|do not|did not)\b/i.test(`${title} ${first}`) ||
!/\brenames Client\.evaluate\(\) to Client\.run\(\)/i.test(first) ||
!/\b(?:deletes|removes|drops) (?:the )?old (?:name|method)\b/i.test(first) ||
!/\b(?:no |without (?:a )?)(?:compatibility )?alias\b/i.test(first)) return [];
}
if (reversedTuples && (approval.test(current) ||
/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:these (?:functions|signatures)|run_eval and run_batch) (?:are (?:now|already) aligned|(?:now )?(?:use|take) the same (?:positional )?order)\b/i.test(current))) return [];
const decision = guardedDeclaration && /^D\s*([1-9]\d*)\s*[—–:-]/i.exec(q.question);
if (decision && new RegExp(`(?:^|[.!?\\n]\\s*)(?:Correction:\\s*)?D\\s*${decision[1]} (?:is|was|has been) (?:withdrawn|rejected|cancelled|canceled|superseded|(?:not|no longer) current)\\b`, 'i').test(current)) return [];
if (guardedDeclaration && /(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:this|that|the) (?:finding|issue|gap|defect|explanation) (?:is|was|has been) (?:cancelled|canceled|superseded)\b/i.test(current)) return [];
if (/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:(?:this|that|the) (?:finding|issue|gap|defect|explanation) (?:is|was|has been) (?:withdrawn|rejected|(?:already )?(?:fixed|resolved)|historical|(?:not|no longer) current|(?:a |only a )?source example)|there is no (?:current )?(?:finding|issue|gap|defect))\b/i.test(current)) return [];
// A declaration's action evidence must belong to a current offered option,
// rather than an example or an explicitly withdrawn correction.
const action = (text: string) => currentProse(text.replace(/`([A-Za-z_$][\w.$/-]*(?:\([^`\n]*\))?)`/g, '$1'), true);
signatureOptions = offered.filter(option => {
const text = `${option.label}\n${option.description ?? ''}`, prose = currentProse(text, true);
if ((upgradeTransition || explainedSubject) && (approval.test(prose) || /(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:do not|don't|never) (?:keep|add|preserve|provide|retain) (?:the |a |an )?(?:compatibility )?alias\b/i.test(prose))) return false;
if (reversedTuples && (approval.test(prose) ||
/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:do not|don't|never) (?:align|unify|standardize|change|make|require) (?:either|both|the|these) (?:functions?|signatures?|arguments?)\b/i.test(prose))) return false;
if (guardedDeclaration && /(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:this|that|the) (?:option|action|correction) (?:is|was|has been) (?:cancelled|canceled|superseded|withdrawn|rejected|(?:not|no longer) current)\b/i.test(prose)) return false;
if (guardedDeclaration && /^(?:Assuming|Provided)\b/im.test(prose)) return false;
if (explainedSubject && /(?:^|[.!?\n]\s*)(?:this|that|the) (?:option|action|correction) (?:applies|belongs) to (?:an? )?(?:other|another|foreign|different) (?:project|SDK|codebase|repository)\b/i.test(prose)) return false;
if (explainedSubject && /(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:(?:do not|don't|never) (?:keep|retain|add|include|emit|provide|forward|call)|(?:this|that|the) (?:option|action|correction) does not (?:keep|retain|add|include|emit|provide|forward|call))\b/i.test(prose)) return false;
if (decision && new RegExp(`(?:^|[.!?\\n]\\s*)(?:Correction:\\s*)?D\\s*${decision[1]} (?:is|was|has been) (?:withdrawn|rejected|cancelled|canceled|superseded|(?:not|no longer) current)\\b`, 'i').test(prose)) return false;
return !/^(?:>|"|“)|^`[^`]*`$/.test(option.label.trim()) && !sourceFrame.test(prose) &&
!/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:this|the) (?:option|action|correction) (?:is|was|has been) (?:withdrawn|rejected|cancelled)\b/i.test(prose);
});
// Preserve inline tuple evidence for the stricter signature parser.
offered = signatureOptions.map(option => ({ ...option, label: action(option.label), description: action(option.description ?? '') }));
if (evidenceJourney) {
const projects = lines.filter(line => /^Project\/branch\/task:/.test(line));
const evidenceFields = lines.filter(line => /^Evidence:/.test(line));
const evidence = evidenceFields[0]?.replace(/^Evidence:\s*/, '') ?? '';
const explain = (lines[explanation] ?? '').replace(/^ELI10:\s*/, '');
const fieldFrame = /^(?:[>"“'‘`]|Source\b|Quoted\b|Historical\b|Earlier\b|Example\b|Hypothetical\b|If\b|Assuming\b|Provided\b)/i;
if (projects.length !== 1 || !/^Project\/branch\/task:\s*EvalKit\b/i.test(projects[0]!) ||
/\b(?:other|another|foreign|different) (?:project|SDK|codebase|repository)\b/i.test(projects[0]!) ||
evidenceFields.length !== 1 || explanation < 2 || lines.indexOf(evidenceFields[0]!) > explanation ||
lines.filter(line => /^ELI10:/.test(line)).length !== 1 || fieldFrame.test(evidence) || fieldFrame.test(explain)) return [];
// Quotes inside a named Evidence field are source data. The title and
// unquoted explanation must independently assert the current problem.
const citations = `${assertionTitle}\n${evidence}`.match(/(?:[\w.-]+\/)*[\w.-]+\.(?:md|txt)\b/g) ?? [];
const allowed = evidenceJourney === 'missing-quickstart'
? ['README.md', 'docs/package-contents.txt', 'package-contents.txt']
: ['README.md', 'docs/current-contracts.md', 'docs/benchmarks.md'];
const required = evidenceJourney === 'missing-quickstart' ? 'docs/package-contents.txt' : 'docs/current-contracts.md';
const ownedStatus = currentProse(q.question.replace(/((?:this|the) (?:evidence|statement) (?:is|was|has been) )["“'‘`](withdrawn|historical|hypothetical|cancelled|canceled|superseded|(?:not|no longer) current)["”'’`]/gi, '$1$2'));
if (/(?:this|the) (?:finding|evidence|explanation|issue) (?:applies|exists|is current) (?:only )?(?:if|when|once) (?:approved|accepted)\b/i.test(current) ||
!citations.includes(required) || !/\bREADME(?:\.md)?\b/.test(evidence) || citations.some(source => !allowed.includes(source)) ||
/\b(?:other|another|foreign|different) (?:project|SDK|codebase|repository)\b/i.test(currentProse(evidence)) ||
/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:this|that|the) (?:evidence|statement) (?:is|was|has been) (?:withdrawn|rejected|historical|hypothetical|cancelled|canceled|superseded|(?:not|no longer) current)\b/i.test(ownedStatus) ||
/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:the )?(?:quickstart|referenced) file (?:is (?:now |already )?(?:shipped|included|present)|now exists)\b/i.test(current) ||
/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:the )?(?:local )?demo (?:no longer|does not|never) (?:waits?|blocks?|requires?)\b/i.test(current)) return [];
if (/\b(?:may|might|could|would|historical|hypothetical|previously|earlier)\b/i.test(assertionTitle) ||
/\b(?:quickstart|README) (?:does not|never|no longer) (?:point|reference)\b/i.test(assertionTitle) ||
/\b(?:that |referenced |quickstart )?file\b[^.\n]{0,30}\b(?:not|never) absent\b/i.test(evidence) ||
/\bfirst local evaluation\b[^.\n]{0,70}\b(?:does not require|never blocks|no longer)\b/i.test(evidence) ||
/\b(?:no longer|does not|never) waits? for (?:that|the) CI check\b/i.test(evidence)) return [];
const asserted = currentProse(explain);
const remedy = offered.some(option => {
const text = `${option.label}\n${option.description ?? ''}`;
if (/\b(?:other|another|foreign|different) (?:project|SDK|codebase|repository|demo|quickstart|file)\b/i.test(text) ||
/\b(?:if|once|when|unless) (?:approved|accepted)|\b(?:after|pending) approval\b/i.test(text) ||
/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:do not|don't|never) (?:change|point|replace|remove|ship|skip|bypass|exempt)\b/i.test(text) ||
/(?:^|[.!?\n]\s*)(?:Correction:\s*)?(?:the )?(?:quickstart still points at the missing file|(?:local )?demo remains gated by CI|README does not point to evalkit\.demo|demo does not (?:skip|bypass) the CI (?:check|gate))\b/i.test(text)) return false;
if (evidenceJourney === 'missing-quickstart') return (
/\b(?:point|replace|make|rewrite)\b[^\n]*\b(?:README|quickstart)\b[^\n]*\b(?:python -m )?evalkit\.demo\b/i.test(text) &&
/\b(?:remove|drop)\b[^\n]*\bfirst_eval\.py\b[^\n]*\breference\b/i.test(text)) ||
/\b(?:ship|add|include)\b[^\n]*\bexamples\/first_eval\.py\b/i.test(text) && /\bwheel\b/i.test(text) && /\bexamples archive\b/i.test(text);
return /\b(?:demo|local(?: mock)?(?: evaluation| eval)?)\b[^.\n]*\b(?:skips?|bypasses?) (?:the )?CI (?:check|gate)\b/i.test(text) ||
/\b(?:exempt|remove|skip|bypass)\b[^.\n]*\b(?:demo|local (?:evaluation|eval|run))\b[^.\n]*\bCI (?:check|gate)\b/i.test(text);
});
const available = assertionTitle.replace(/\bisn['’]t\b/gi, 'is not').replace(/\bdoesn['’]t\b/gi, 'does not');
if (evidenceJourney === 'missing-quickstart') return (
/\b(?:points?|references?)\b[^.?!]*\b(?:file|example)\b[^.?!]*\b(?:not (?:shipped|included|packaged)|does not (?:ship|exist)|absent|missing)\b/i.test(available) &&
/\bexamples\/first_eval\.py\b/.test(evidence) && /\bpython -m evalkit\.demo\b/.test(evidence) &&
/\b(?:that |referenced |quickstart )?file\b[^.\n]{0,30}\babsent\b[^.\n]*\b(?:package|wheel)\b[^.\n]*\bexamples archive\b/i.test(evidence) &&
/\b(?:command|quickstart)\b[^.?!]*\bfails?\b[^.?!]*\b(?:file-not-found|missing file)\b/i.test(asserted) && remedy
) ? ['missing-quickstart'] : [];
return /\b(?:mandatory|required|blocks?|waits?)\b/i.test(assertionTitle) &&
/\bfirst local evaluation\b/i.test(evidence) && /\b(?:requires?|blocks?|waits?)\b/i.test(evidence) && /\b(?:five minutes|5.minutes|300s)\b/i.test(evidence) &&
/\bdemo\b/i.test(asserted) && /\bmock transport\b/i.test(asserted) && /\bwait\b/i.test(asserted) && remedy
? ['local-ci-gate'] : [];
}
}
const options = offered.map(o => `${o.label} ${o.description ?? ''}`);
// New explained questions require one complete current correction. Labels
// cannot lend a missing cause, warning or migration path to another option.
const explainedRemedy = offered.some(option => {
const text = option.description ?? '';
const action = text.split(/\n[✅❌]/)[0]!;
if (/\b(?:if|unless|assuming|provided|historical|hypothetical)\b/i.test(action) ||
/\b(?:other|another|foreign|different) (?:project|SDK|codebase|repository)\b/i.test(action) ||
/(?:^|[.!?\n]\s*)(?:do not|don't|never) (?:keep|retain|add|include|emit|provide|forward|call)\b/i.test(action)) return false;
if (explainedAuthentication) return (
/^(?:(?:Keep|Retain) the AuthError class\.\s*Message|AuthError) (?:gains|includes)\b[^.]*\bcode\b[^.]*\bcause\b[^.]*\bfix\b/i.test(action) ||
/^Stable code EVALKIT_AUTH_INVALID_KEY,\s*cause,\s*(?:console URL )?fix\b/i.test(action)) &&
!/\b(?:no|without|not)\b[^.]*\b(?:code|cause|fix)\b/i.test(action);
return (/^(?:Client\.)?evaluate\(\) (?:stays|remains) as (?:a |an )?(?:thin |deprecated |compatibility )?(?:wrapper|alias) that (?:calls|forwards to) Client\.run\(\) and emits DeprecationWarning\b/i.test(action) ||
/^evaluate\(\) delegates to run\(\) with DeprecationWarning naming run\(\) and (?:3\.0|v3)\b/i.test(action)) &&
/\bmigration (?:guide|section)\b/i.test(action) && !/\b(?:no|without|not)\b[^.]*\b(?:alias|wrapper|warning|migration (?:guide|section))\b/i.test(action);
});
const ownUpgradeAlias = (option: string) => !(upgradeTransition || vanishingUpgrade) || (
/(?<![\w.])(?:Client\.)?evaluate\(\) (?:stays|remains) (?:as )?(?:a |an )?(?:thin |deprecated |compatibility )?alias\b|\b(?:keep|retain|preserve) (?<![\w.])(?:Client\.)?evaluate\(\) as (?:a |an )?(?:deprecated |compatibility )?alias\b/i.test(option) &&
!/\b(?:no |without (?:a )?)(?:compatibility )?alias\b|\b(?:do not|don't|never) (?:keep|retain|preserve) (?:Client\.)?evaluate\(\)/i.test(option));
const labels = q.options.map(o => o.label.trim().replace(/\s*\(recommended\)$/i, '').toLowerCase());
const yesNo = labels.length === 2 && labels.includes('yes') && labels.includes('no');
// A terse Yes/No panel still resolves an action explicitly asked in the
// main question; action words in background prose never supply this arm.
const directAction = (verbs: string) => yesNo && new RegExp(
`^(?:should|shall|can|do|would) (?:we|I) (?:${verbs})\\b[^?]*\\?$`, 'i').test(title);
const found: DevexSeededGap[] = [];
if (/\bCI\b/i.test(title) && /\b(?:local|demo|first)\b/i.test(title) &&
/\b(?:gate|check|blocks?|waits?|bypass|mandatory|required)\b/i.test(title) &&
(options.some(o => /\b(?:no CI gate|remove|move|skip|bypass|gate)\b/i.test(o) && /\b(?:CI|check|gate|local|demo)\b/i.test(o)) || directAction('remove|move|skip|bypass|gate'))) found.push('local-ci-gate');
if (/\bquickstart\b|examples\/first_eval\.py/i.test(title) &&
/\b(?:README|file|example|demo|missing|absent|package|wheel|ship|point)\b|first_eval\.py/i.test(title) &&
(!declaration || absentReference || /\b(?:not (?:shipped|included|available|present)|missing|absent|nonexistent|does not exist)\b/i.test(title)) &&
(options.some(o => /\b(?:point|ship|add|demo is)\b/i.test(o) && /\bquickstart\b|first_eval\.py/i.test(o)) || directAction('point|ship|add|replace|fix'))) found.push('missing-quickstart');
if (explainedReversedSignatures({ ...q, options: signatureOptions }, title) || (/\brun_eval\b/i.test(title) && /\brun_batch\b/i.test(title) &&
/\b(?:arguments?|order|positional|reversed|opposite|consistent|align|unify|dataset|evaluator)\b/i.test(title) &&
(!declaration || reversedTuples || /\b(?:reversed|opposite|swapped|inconsistent)\b/i.test(title)) &&
(options.some(o => (!reversedTuples || /\bboth functions\b|\brun_eval\b[^\n]*\brun_batch\b/i.test(o)) &&
/\b(?:align|unify|standardize|keyword|swap)\b/i.test(o) && /\b(?:order|dataset|arguments?|positional)\b/i.test(o)) || directAction('align|unify|standardize|enforce|make')))) found.push('reversed-arguments');
if ((explainedAuthentication || opaqueAuthentication || /\bAuthError\b|\binvalid API key\b/i.test(title)) &&
(explainedAuthentication || /\b(?:error|message|code|cause|fix|guidance|opaque|explain)\b|request failed/i.test(title)) &&
(!declaration || opaqueAuthentication || /\b(?:no (?:cause|fix|explanation|code)|opaque)\b|request failed/i.test(title)) &&
(explainedAuthentication ? explainedRemedy : (options.some(o => (!opaqueAuthentication || /\bAuthError\b/i.test(o)) && (/\bcodes?\b/i.test(o) || /^(?:[A-D]\)\s*)?Coded\b/i.test(o)) && /\b(?:cause|fix|link)\b/i.test(o)) || directAction('add|include|explain|replace|report|give')))) found.push('opaque-auth-error');
if (/Client\.evaluate\b/i.test(title) &&
/Client\.run\b|\b(?:v\d+|version \d+|alias|deprecation|migration)\b/i.test(title) &&
(upgradeVocabulary || upgradeTransition || explainedUpgrade) &&
(!declaration || /\b(?:no |without (?:a )?)(?:compatibility )?(?:alias|warning|migration (?:guide|path))\b/i.test(title)) &&
(explainedUpgrade ? explainedRemedy : (options.some(o => ownUpgradeAlias(o) && /\balias\b/i.test(o) && /\b(?:warning|DeprecationWarning|migration)\b/i.test(o)) || directAction('keep|add|preserve|provide|retain')))) found.push('breaking-upgrade');
return found;
}
/** Extra real decisions are permitted; each seeded gap needs its own completed native call. */
export function devexSeedCoverage(transcript: PlanCountTranscript) {
const decisions = Object.fromEntries(DEVEX_SEEDED_GAPS.map(gap => [gap, []])) as Record<DevexSeededGap, string[]>;
const batched: string[] = [];
const invalid: string[] = [];
const sessions = new Set(transcript.calls.map(c => c.sessionId));
if (transcript.status !== 'ready' || sessions.size !== 1 || sessions.has('')) invalid.push('missing or mixed native session');
const ids = new Set<string>();
for (const call of transcript.calls) {
const id = `${call.sessionId}:${call.toolUseId}`;
if (!call.toolUseId || ids.has(id)) { invalid.push(`missing or repeated native call: ${id}`); continue; }
ids.add(id);
const gaps = call.questions.flatMap(decisionGaps);
if (!gaps.length) continue;
if (call.questions.length !== 1 || gaps.length !== 1 || call.questions[0]!.multiSelect) {
batched.push(id); continue;
}
const q = call.questions[0]!;
const labels = q.options.map(o => o.label);
const complete = call.answered === true && call.failed === false &&
Array.isArray(call.unansweredQuestionIndices) && call.unansweredQuestionIndices.length === 0 &&
Number.isFinite(Date.parse(call.answeredAt ?? '')) && q.options.length >= 2 && q.options.length <= 4 &&
new Set(labels).size === labels.length && labels.every(Boolean) &&
Object.keys(call.answers ?? {}).length === 1 && labels.includes(call.answers?.[q.question] ?? '');
if (complete) decisions[gaps[0]!]!.push(id);
}
const missing = DEVEX_SEEDED_GAPS.filter(gap => decisions[gap].length === 0);
const matchedIds = new Set(Object.values(decisions).flat());
return {
complete: invalid.length === 0 && batched.length === 0 && missing.length === 0 && matchedIds.size >= DEVEX_SEEDED_GAPS.length,
missing, decisions, batched, invalid,
};
}
+1 -20
View File
@@ -6,7 +6,7 @@
*/
import '../../lib/conductor-env-shim';
import { describe, test, beforeAll, afterAll, expect } from 'bun:test';
import { describe, test, afterAll, expect } from 'bun:test';
import type { SkillTestResult } from './session-runner';
import { EvalCollector, judgePassed } from './eval-store';
import type { EvalTestEntry } from './eval-store';
@@ -14,7 +14,6 @@ import { judgeRecommendation, type RecommendationScore } from './llm-judge';
import { selectTests, detectBaseBranch, getChangedFiles, E2E_TOUCHFILES, E2E_TIERS, GLOBAL_TOUCHFILES } from './touchfiles';
import { WorktreeManager } from '../../lib/worktree';
import type { HarvestResult } from '../../lib/worktree';
import { spawnSync } from 'child_process';
import { preflightAnthropicApi } from './anthropic-preflight';
import * as fs from 'fs';
import * as path from 'path';
@@ -400,24 +399,6 @@ export function harvestAndCleanup(testName: string): HarvestResult | null {
return result;
}
/**
* Convenience: describe block with automatic worktree isolation + harvest.
* Any test file can use this to get real repo context instead of a tmpdir.
* Note: tests with planted-bug fixtures should NOT use this — they need their fixture repos.
*/
export function describeWithWorktree(
name: string,
testNames: string[],
fn: (getWorktreePath: () => string) => void,
) {
describeIfSelected(name, testNames, () => {
let worktreePath: string;
beforeAll(() => { worktreePath = createTestWorktree(name); });
afterAll(() => { harvestAndCleanup(name); });
fn(() => worktreePath);
});
}
export { judgePassed } from './eval-store';
export { EvalCollector } from './eval-store';
export type { EvalTestEntry } from './eval-store';
File diff suppressed because it is too large. Load diff
-96
View File
@@ -1,96 +0,0 @@
import * as fs from 'node:fs';
import * as path from 'node:path';
import type { AskUserQuestionFingerprint } from './claude-pty-runner';
import type { NativeQuestion } from './plan-skill-questions';
type Commitment = Readonly<{ id: string; label: string; description: string }>;
/** Author-owned offered choices. Exact native label AND description establish
* authority; arbitrary question prose and recommendations never enlarge it.
* This is an explicit fixture interface, not a natural-language classifier. */
export const ENG_COUNT_COMMITMENTS: readonly Commitment[] = Object.freeze([
{ id: 'keep-seeded-scope', label: 'Keep seeded scope',
description: 'Keep only the implementation scope already proposed in PLAN.md. The review must still address every seeded issue. ✅ Retains the requested refactor. ❌ Does not remove its existing defects. No additional feature or product behavior is authorized; explain and record this scope choice.' },
{ id: 'keep-classes', label: 'Keep five classes',
description: 'Keep AuthBroker, TokenStore, SessionMint, AuthCache and RequestPolicy in the proposed refactor. Explain their responsibilities and tradeoff in the review. ✅ Retains the proposed class scope. ❌ Does not reduce its complexity. No new feature, service or product behavior is authorized.' },
{ id: 'reduce-classes', label: 'Simplify class layout',
description: 'Consolidate responsibilities only among the five proposed classes and the existing cache adapter. Explain the resulting arrangement in the review. ✅ Reduces structural overhead. ❌ Requires changing the proposed boundaries. Preserve existing product behavior; no new responsibility or feature is authorized.' },
{ id: 'cache-ownership', label: 'Own existing cache',
description: 'Authorize explicit dependency ownership and isolated mutation responsibility for the existing AuthCache. Explain the proposed ownership and its tests in the review. ✅ Addresses module-level mutable sharing. ❌ Requires wiring changes. Retain the existing adapter, key, validity and invalidation contracts; no new storage or cross-request coordination is authorized.' },
{ id: 'error-handling', label: 'Handle existing errors',
description: 'Authorize explicit handling or propagation of the existing failure classes in validateAndDispatch. Explain the mapping and its tests in the review. ✅ Makes hidden failures explicit. ❌ Requires caller and test review. Preserve the existing observable outcome contract; no new product policy, network behavior, retry or timeout is authorized.' },
{ id: 'parallel-idp', label: 'Parallelize five calls',
description: 'Authorize concurrent execution of only the five existing independent IDP calls within each request. Explain independence and error behavior in the review. ✅ Removes their sequential wait. ❌ Requires concurrency tests. Preserve the existing success and failure policy; no extra calls, cross-request deduplication, new timeout or retry is authorized.' },
{ id: 'tests-only', label: 'Tests only',
description: 'Authorize adding or updating tests only for this plan and already-authorized changes. The review must identify and explain the concrete test obligations. ✅ Adds evidence. ❌ Authorizes no production implementation or behavior change. Record this exact answer without treating any unperformed test as passed.' },
{ id: 'document-only', label: 'Document only',
description: 'Authorize documenting the existing component contract, evidence gap or risk described in this review. ✅ Makes the current constraint explicit. ❌ Does not implement a remedy. No new product behavior or implementation permission is granted; retain any unresolved work honestly.' },
{ id: 'retain-behavior', label: 'Keep current behavior',
description: 'Decline the proposed additional behavior and keep the existing behavior for this decision. ✅ Keeps this refactor within its stated scope. ❌ Leaves the additional risk or opportunity unchanged. The review may report that finding; no new state, network call, product policy or implementation is authorized.' },
{ id: 'defer', label: 'Defer implementation',
description: 'Defer the proposed implementation to separate future work. ✅ Records the issue without expanding this refactor. ❌ Leaves its implementation unresolved. Explain the deferred work in the review; this grants no permission to start it and no completed-work credit.' },
{ id: 'finish', label: 'Finish; follow up later',
description: 'Finish this engineering review after its required report and completion steps. I will handle later reviews and implementation manually. ✅ Keeps this invocation bounded. ❌ Provides no follow-up review coverage. No implementation or additional skill invocation is authorized.' },
{ id: 'continue', label: 'Continue this review',
description: 'Continue only the remaining required work of this engineering review. ✅ Allows its current obligations to finish. ❌ Does not authorize another review or implementation. Existing scope, permissions and unresolved conditions remain unchanged.' },
].map(row => Object.freeze(row)));
const ACTOR_INSTRUCTIONS = [
'## Declared review actor interface',
'The plan above remains the review target, including every seeded issue. Do not change its findings, prescribe a question count/order, or treat this interface as test evidence.',
'For each native question, offer exactly one applicable author-owned choice from the catalog below, alongside your own alternatives. Copy that choice’s label and description exactly, with no recommendation tag, preview or additional option fields. Other offered choices remain unapproved and may be freely worded.',
'The actor selects only that complete copied commitment. Put the question-specific explanation, recommendation, pros/cons and comparisons in the question text; none can add authority to the copied commitment. A recommendation for a different option remains visible and is not approved.',
'Zero or multiple catalog matches, modified or spoofed commitments, and duplicate labels for the selected choice stop this fixture without an answer. Reordering intact options does not change which commitment is authorized.',
'The catalog does not require an item to be offered or supply any missing review analysis. The review must discover and justify its own test obligations. Actual answers and the complete report remain required.',
'```json',
JSON.stringify(ENG_COUNT_COMMITMENTS.map(({ label, description }) => ({ label, description })), null, 2),
'```',
].join('\n');
export function engCountActorRequest(seed: string): string {
if (!seed.trim() || seed.includes('## Declared review actor interface')) throw new Error('Eng actor requires one original seed');
return `${seed}\n\n${ACTOR_INSTRUCTIONS}`;
}
/** No normalization: whitespace, punctuation, labels and descriptions belong to
* this explicit interface. Reordering intact choices preserves their intent. */
export function pickEngCountQuestion(question: NativeQuestion): number {
if (!question.header?.trim() || !question.question?.trim() || question.multiSelect ||
question.options.length < 2 || question.options.length > 4)
throw new Error('Eng actor requires one complete single-choice question');
const matches = question.options.flatMap((option, index) =>
ENG_COUNT_COMMITMENTS.some(row => row.label === option.label && row.description === option.description) ? [index] : []);
if (matches.length !== 1) throw new Error('Eng actor requires exactly one complete author-owned offered commitment');
const index = matches[0]!, selected = question.options[index]!;
if (Object.keys(selected).some(key => key !== 'label' && key !== 'description') ||
question.options.filter(option => option.label === selected.label).length !== 1)
throw new Error('Eng actor received a modified or ambiguous selected commitment');
return index + 1;
}
/** Bind the declared request to the actual isolated seed and the existing
* runner's complete current native active-tab capture. No UI-only fallback. */
export function createEngCountActor(request: string) {
if (!request.endsWith(`\n\n${ACTOR_INSTRUCTIONS}`)) throw new Error('Eng actor request lacks its declared catalog');
let session: string | undefined;
return (_routing: AskUserQuestionFingerprint, active: AskUserQuestionFingerprint,
context: Readonly<{ cwd: string; deadlineAt: number }>): number => {
const file = path.join(context.cwd, 'PLAN.md'), stat = fs.lstatSync(file);
if (!stat.isFile() || stat.isSymbolicLink() || fs.readFileSync(file, 'utf8') !== request ||
!Number.isFinite(context.deadlineAt) || context.deadlineAt <= Date.now())
throw new Error('Eng actor requires the current owned request and original deadline');
const call = active.nativeCall, index = active.nativeQuestionIndex;
const question = index === undefined ? undefined : call?.questions[index];
const signature = call && `${call.sessionId}:${call.toolUseId}`;
if (!call || !call.sessionId || !call.toolUseId || call.answered || call.failed || !question ||
(session !== undefined && session !== call.sessionId) || call.answers?.[question.question] !== undefined ||
active.signature !== (call.questions.length === 1 ? signature : `${signature}:question:${index}`) ||
active.promptSnippet !== `${question.header} ${question.question}` ||
active.options.length !== question.options.length || !active.options.every((option, i) =>
option.index === i + 1 && option.label === question.options[i]!.label))
throw new Error('Eng actor requires the complete matched pending native tab');
const chosen = pickEngCountQuestion(question);
session ??= call.sessionId;
return chosen;
};
}
-60
View File
@@ -1,60 +0,0 @@
import { execFileSync } from 'node:child_process';
import * as fs from 'node:fs';
import * as path from 'node:path';
import { seedPlanReviewProject } from './ceo-finding-fixture';
/** Supply the existing behavior that the plan says it will rewrite. The new
* AuthBroker/SessionMint/cache implementation remains proposed and absent. */
export function seedEngFindingProject(projectDir: string, plan: string): string {
const input = plan + '\n\n## Existing behavior and boundaries\n' + [
'Read `src/legacy-auth.ts`: this is the existing flow, not the proposed refactor.',
'The HTTP admission middleware, IDP policy client, and session service keep their',
'documented contracts. The five policy names identify the existing independent',
'read-only verdict calls made by this function.',
'For this synthetic fixture, the unchanged platform admission limit and IDP quota',
'reserve concurrency and rate capacity for five policy calls per admitted request,',
'including denied requests. This refactor does not require a capacity rollout.',
'Preserve the return value, all three public failure codes, and which outcome',
'wins: evaluate failures in the existing POLICIES order, not response-arrival order.',
'No session lifecycle, HTTP protocol, schema, or deployment change is proposed.',
'Those adapter implementations belong to the existing platform outside this',
'fixture; their interface here defines the boundary of this refactor.',
'The existing deployment system restores the prior build artifact for rollback;',
'there is no per-function rollout flag in this module or deployment change here.',
'This is a Bun TypeScript package. Use its existing `bun test` command for new',
'tests; choosing a different runner or adding a toolchain is outside the refactor.',
'',
'## Proposed class roles',
'The four classes are a proposed organization of operations and values already',
'present in `legacyAuthFlow`, not additional product features or implemented code:',
'- AuthBroker coordinates the policy checks and the existing session call.',
'- SessionMint delegates to `platform.issueSession(identity)`; the platform still',
' owns session IDs, expiry, revocation, and storage.',
'- TokenStore is a request-local holder for the existing Identity and returned',
' Session values; it adds no persistence, token issuance, or renewal.',
'- RequestPolicy wraps one existing POLICIES entry and its `checkPolicy` call;',
' it adds no policy, configuration, or precedence rule.',
'This class arrangement is still a proposal to review, not an accepted design.',
'The separately proposed shared mutable AuthCache remains in the plan.',
'',
'## Undecided internal failure interface',
// Newly authored synthetic input; this is not evidence about prior reviews.
'The proposed validateAndDispatch catches would swallow failures. The refactor',
'has not chosen how those failures reach the existing public auth boundary:',
'typed exception propagation or a discriminated result handled exhaustively',
'before that boundary.',
'Both must preserve the public failure codes, causes and POLICIES-order precedence.',
'This internal choice fits the accepted single-function/module or class organization;',
'it does not require a new helper or adapter. It is independent of sequential or',
'parallel IDP dispatch and the algorithm used to settle the policy outcomes.',
].join('\n') + '\n';
seedPlanReviewProject(projectDir, input, 'plan-eng-review');
fs.mkdirSync(path.join(projectDir, 'src'));
fs.copyFileSync(path.resolve(import.meta.dir, '../fixtures/eng-existing-auth/legacy-auth.ts'), path.join(projectDir, 'src/legacy-auth.ts'));
fs.copyFileSync(path.resolve(import.meta.dir, '../fixtures/eng-existing-auth/package.json'), path.join(projectDir, 'package.json'));
const git = (...args: string[]) => execFileSync('git', args, { cwd: projectDir, stdio: 'pipe', timeout: 10_000 });
git('add', 'src/legacy-auth.ts', 'package.json');
git('-c', 'user.name=Finding fixture', '-c', 'user.email=fixture@gstack.test', 'commit', '-m', 'Supply existing auth behavior');
git('update-ref', 'refs/remotes/origin/main', 'HEAD');
return input;
}
-67
View File
@@ -1,67 +0,0 @@
/** A recorded legacy oracle can survive removal of the implementation it sampled. */
export function hasRetainedLegacyCorpus(
current: ReadonlyArray<{ title: string; body: string[] }>, snapshot: string,
): boolean {
const flat = (s: string) => s.replace(/\s+/g, ' ').trim();
const quoted = (s: string) => s.replace(/"[^"\n]*"|“[^”\n]*”|(?<![\w])'[^'\n]*'(?![\w])|‘[^’\n]*’/g, '');
const source = (s: string) => /(?:^|\n)\s*(?:source(?: excerpt| material)?|quoted(?: source)?|historical(?: example| assessment)?|if approved|once approved|when approved|pending approval|assuming approval|provided approval)\s*[,.:—-]/i.test(quoted(s));
const status = '(?:withdrawn|rejected|declined|cancelled|canceled|superseded|deferred|optional|proposed|not current|no longer current|not required|no longer required|conditional on approval)';
const owner = '(?:(?:this|the) (?:(?:legacy|recorded|baseline) )?(?:(?:regression|parity) )?(?:suite|test|requirement|verification|oracle|corpus))';
const scalar = (s: string, id: string) => quoted(s.replace(new RegExp(
`((?:^|[.!?;]\\s+|\\n)\\s*(?:Correction:\\s*)?(?:${id}(?: verification)?|${owner}) (?:is|are|was|were|has been|have been) )["“'‘](${status})["”'’]`, 'gim'), '$1$2'));
const inactive = (s: string, id: string) => source(s) || new RegExp(
`\\b(?:${id}(?: verification)?|${owner}) (?:is|are|was|were|has been|have been) ${status}\\b|\\b(?:only if|unless|pending) (?:user )?approv`, 'i').test(scalar(s, id));
const negated = (s: string) => /\b(?:do not|don't|never|skip|omit|defer|cancel) (?:add|write|implement|record|capture|pin|run|assert|retain|keep)\b/i.test(quoted(s));
const owned = (s: string) => !source(s) && !negated(s) && snapshot.includes(flat(s));
for (const declaration of current) {
if (!/^CRITICAL regression test \([^)]*\bmandatory\b[^)]*\)$/i.test(declaration.title)
|| /\b(?:not|never|no longer) mandatory\b/i.test(declaration.title)) continue;
const body = declaration.body.join('\n').trim();
const add = /^Add ([A-Za-z][\w/.-]*\.test\.[jt]s):$/m.exec(body);
if (!add || !owned(body) || inactive(body, 'T[1-9]\\d*')) continue;
const bullets = body.split(/\n(?=- )/).slice(1).map(flat);
const capture = bullets.findIndex(s => /^- (?:Record|Capture|Pin) a corpus of .+ with the legacy decision for each\.$/i.test(s));
const parity = bullets.findIndex(s => /^- Run the (?:same )?corpus through the new [A-Za-z][\w]* path and assert identical allow\/deny and reason code for every entry\.$/i.test(s));
if (capture < 0 || parity <= capture || !bullets.some(s => /^- This test is also the shadow-mode oracle; it stays after legacy deletion, re-pointed at the recorded decisions\.$/i.test(s))) continue;
// Retention makes the old behavior the oracle. Building independent new
// modules before recording it does not itself rewrite that old behavior.
const retention = current.filter(s => s.title === 'What already exists').flatMap(s => s.body.join('\n').split(/\n(?=- )/))
.find(s => /^- legacyAuthFlow\(\): retained behind the per-tenant flag as the shadow oracle and regression baseline until deletion\.$/i.test(flat(s)) && owned(s) && !inactive(s, 'T[1-9]\\d*'));
if (!retention) continue;
for (const section of current.filter(s => s.title === 'Implementation Tasks')) {
const taskBody = section.body.join('\n').trim();
const tasks = taskBody.split(/\n(?=- )/);
const rows = tasks.map(body => ({ body, match: /^- (?:\[[ xX]\] )?(T[1-9]\d*)(?: \([^\n)]*\))? [—–-] (.+)(?:\n|$)/.exec(body) })).filter(row => row.match);
if (rows.length !== new Set(rows.map(row => row.match![1])).size) continue;
for (const row of rows) {
const id = row.match![1]!, title = row.match![2]!;
const action = row.body.replace(/^ - Surfaced by:.*$/gm, '');
const preceding = taskBody.slice(0, taskBody.indexOf(row.body)).trim().split('\n').at(-1) ?? '';
const files = [...row.body.matchAll(/^ - Files: ([^\n]+)$/gm)];
const verifies = [...row.body.matchAll(/^ - Verify: ([^\n]+)$/gm)];
if (!/^[A-Za-z][\w -]* [—–-] CRITICAL: recorded-corpus parity test legacy vs [A-Za-z][\w]*$/i.test(title)
|| files.length !== 1 || verifies.length !== 1 || !snapshot.includes(flat(row.body)) || source(action) || negated(action) || source(preceding)
|| inactive(action, id) || !files[0]![1]!.split(',').map(s => s.trim()).includes(add[1]!)) continue;
if (!/^100% decision \+ reason-code parity across the corpus$/i.test(verifies[0]![1]!)) continue;
const cancelled = current.some(s => {
if (/\b(?:history|historical|source|quoted|example)\b/i.test(s.title)) return false;
const raw = s.body.join('\n'), value = scalar(raw, id);
const named = /^(.*?)\b(?:regression|characterization|parity)\s+(?:suite|test)/i.exec(s.title)?.[1]?.trim();
const foreign = Boolean(named && !/^(?:(?:critical|recorded|required|current|final|updated)\s*)*(?:legacy(?:AuthFlow\(\))?)?[\s:—-]*$/i.test(named));
if (new RegExp(`^\\s*\\|\\s*${id}\\s*\\|\\s*["“'‘]?${status}["”'’]?\\s*\\|`, 'im').test(raw)) return true;
return value.split(/\n|[.!?;]\s+/).some(line => {
if (source(line) || /^\s*(?:if|unless|assuming|provided)\b/i.test(line)) return false;
if (foreign && !new RegExp(`\\b${id}\\b|legacyAuthFlow|\\blegacy (?:regression|parity) (?:suite|test|corpus|baseline)\\b`, 'i').test(line)) return false;
return inactive(line, id) || new RegExp(`^\\s*(?:Correction:\\s*)?legacyAuthFlow\\(\\) (?:is|was|has been|will be) (?:modified|changed|rewritten|refactored|removed|deleted) before ${id}\\b`, 'i').test(line)
|| /^\s*(?:Correction:\s*)?legacyAuthFlow\(\) (?:is|was|has been|will be) (?:modified|changed|rewritten|refactored|removed|deleted) before (?:the )?(?:(?:legacy|recorded|regression) )?(?:corpus|baseline)(?: is)? (?:recorded|captured|pinned)\b/i.test(line)
|| /^\s*(?:the|this) (?:(?:legacy|regression) )?(?:corpus|baseline) is (?:recorded|captured|pinned) (?:only )?after legacyAuthFlow\(\) is (?:modified|changed|rewritten|refactored|removed|deleted)\b/i.test(line)
|| /\b(?:the|this) legacy (?:regression baseline|shadow oracle) (?:is|was|has been) (?:changed|rewritten|removed|deleted|replaced)\b/i.test(line);
});
});
if (!cancelled) return true;
}
}
}
return false;
}
File diff suppressed because it is too large. Load diff
+13 -36
View File
@@ -43,44 +43,24 @@ export const ALL_TIERS = {
PTY_LONG_MS,
} as const;
/**
* Explicit exception for one uninterrupted four-phase workflow. These are
* specified allowances, not measured latency or a conservative confidence bound.
* The historical 900-second failure remains a failure. Ordinary tiers do not grow.
*/
export const AUTOPLAN_CHAIN_BUDGET = {
id: 'autoplan-four-native-phases-v1',
file: 'test/skill-e2e-autoplan-chain.test.ts',
workMs: 4 * PTY_LONG_MS,
sessionMs: 84 * 60_000,
testMs: 85 * 60_000,
shardMs: 172 * 60_000,
retries: 1,
shardReserveMs: 2 * 60_000,
ciJobMs: 200 * 60_000,
ciReserveMs: 28 * 60_000,
reason: 'One command must complete CEO, Design, DX and Eng, including native reviews and amendment handoffs.',
} as const;
/** Supervision reserve added to every registered whole-file wall. */
export const SHARD_RESERVE_MS = 2 * 60_000;
/** Whole-file supervision must cover each existing attempt and its retry.
* These six fixtures already allow 25 minutes per case; the old 30-minute
* These fixtures already allow 25 minutes per case; the old 30-minute
* wall could kill a second attempt after five minutes. No case budget grows.
* Reserve the sequential upper bound even when Bun runs sibling cases together.
*/
export const FINDING_RETRY_BUDGETS = [
{ file: 'test/skill-e2e-plan-ceo-finding-count.test.ts', cases: 2 },
{ file: 'test/skill-e2e-plan-ceo-split-overflow.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-design-finding-count.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-devex-finding-count.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-eng-finding-count.test.ts', cases: 1 },
{ file: 'test/skill-e2e-plan-eng-multi-finding-batching.test.ts', cases: 1 },
].map(({ file, cases }) => ({
file, cases,
id: `${file.slice('test/skill-e2e-'.length, -'.test.ts'.length)}-existing-retry-v1`,
testMs: 1_500_000,
retries: 1,
shardReserveMs: AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
shardMs: cases * 1_500_000 * 2 + AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: cases * 1_500_000 * 2 + SHARD_RESERVE_MS,
}));
/** Three existing captures and one configured retry; only supervision grows. */
@@ -90,8 +70,8 @@ export const AUQ_CONSISTENCY_RETRY_BUDGET = {
cases: 1,
testMs: 3 * CAPTURE_MS + 60_000,
retries: 1,
shardReserveMs: AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
shardMs: (3 * CAPTURE_MS + 60_000) * 2 + AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: (3 * CAPTURE_MS + 60_000) * 2 + SHARD_RESERVE_MS,
} as const;
/** These fixtures have a fixed case count in every supported tier. */
@@ -109,9 +89,8 @@ export const FILE_RETRY_BUDGETS = [
{ file: 'test/skill-e2e-shared-libs-paths.test.ts', attemptMs: 3 * CAPTURE_LONG_MS, retries: 1 },
{ file: 'test/skill-e2e-ship-docsync.test.ts', attemptMs: 5 * CAPTURE_LONG_MS + 8 * CAPTURE_MS, retries: 1 },
// Seventeen workflow judges include their 10s recording grace; the other
// eleven judges retain 120s. Supervise all 28 and the existing one retry.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 11 * JUDGE_MS, retries: 1 },
{ file: 'test/codex-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_LONG_MS + 10_000), retries: 1 },
// seven judges retain 120s. Supervise all 24 and the existing one retry.
{ file: 'test/skill-llm-eval.test.ts', attemptMs: 17 * (JUDGE_MS + 10_000) + 7 * JUDGE_MS, retries: 1 },
{ file: 'test/skill-e2e-auq-matrix.test.ts', attemptMs: 6 * CAPTURE_MS, retries: 1 },
{ file: 'test/skill-e2e-plan-format.test.ts', attemptMs: 4 * (CAPTURE_MS + 10_000), retries: 1 },
{ file: 'test/skill-e2e-auto-decide-preserved.test.ts', attemptMs: PTY_MS, retries: 1 },
@@ -128,16 +107,14 @@ export const FILE_RETRY_BUDGETS = [
].map(({ file, attemptMs, retries }) => ({
file, attemptMs, retries,
id: `${file.slice('test/'.length, -'.test.ts'.length)}-existing-retry-v1`,
shardReserveMs: AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
shardMs: attemptMs * (retries + 1) + AUTOPLAN_CHAIN_BUDGET.shardReserveMs,
shardReserveMs: SHARD_RESERVE_MS,
shardMs: attemptMs * (retries + 1) + SHARD_RESERVE_MS,
})),
];
/** The only registered over-tier test budget; arbitrary per-file escapes fail. */
/** No paid test may exceed the ordinary tiers; arbitrary per-file escapes fail. */
export function assertPaidTestBudget(file: string, ms: number): void {
if (!Number.isSafeInteger(ms) || ms <= 0 ||
(ms > PTY_LONG_MS * 1.25 &&
(file !== AUTOPLAN_CHAIN_BUDGET.file || ms !== AUTOPLAN_CHAIN_BUDGET.testMs))) {
if (!Number.isSafeInteger(ms) || ms <= 0 || ms > PTY_LONG_MS * 1.25) {
throw new Error(`Unregistered paid test budget: ${file}: ${ms}`);
}
}
-104
View File
@@ -1,104 +0,0 @@
import { describe, test, expect } from 'bun:test';
import { parseGeminiJSONL } from './gemini-session-runner';
// Fixture: actual Gemini CLI stream-json output with tool use
const FIXTURE_LINES = [
'{"type":"init","timestamp":"2026-03-20T15:14:46.455Z","session_id":"test-session-123","model":"auto-gemini-3"}',
'{"type":"message","timestamp":"2026-03-20T15:14:46.456Z","role":"user","content":"list the files"}',
'{"type":"message","timestamp":"2026-03-20T15:14:49.650Z","role":"assistant","content":"I will list the files.","delta":true}',
'{"type":"tool_use","timestamp":"2026-03-20T15:14:49.690Z","tool_name":"run_shell_command","tool_id":"cmd_1","parameters":{"command":"ls"}}',
'{"type":"tool_result","timestamp":"2026-03-20T15:14:49.931Z","tool_id":"cmd_1","status":"success","output":"file1.ts\\nfile2.ts"}',
'{"type":"message","timestamp":"2026-03-20T15:14:51.945Z","role":"assistant","content":"Here are the files.","delta":true}',
'{"type":"result","timestamp":"2026-03-20T15:14:52.030Z","status":"success","stats":{"total_tokens":27147,"input_tokens":26928,"output_tokens":87,"cached":0,"duration_ms":5575,"tool_calls":1}}',
];
describe('parseGeminiJSONL', () => {
test('extracts session ID from init event', () => {
const parsed = parseGeminiJSONL(FIXTURE_LINES);
expect(parsed.sessionId).toBe('test-session-123');
});
test('concatenates assistant message deltas into output', () => {
const parsed = parseGeminiJSONL(FIXTURE_LINES);
expect(parsed.output).toBe('I will list the files.Here are the files.');
});
test('ignores user messages', () => {
const lines = [
'{"type":"message","role":"user","content":"this should be ignored"}',
'{"type":"message","role":"assistant","content":"this should be kept","delta":true}',
];
const parsed = parseGeminiJSONL(lines);
expect(parsed.output).toBe('this should be kept');
});
test('extracts tool names from tool_use events', () => {
const parsed = parseGeminiJSONL(FIXTURE_LINES);
expect(parsed.toolCalls).toHaveLength(1);
expect(parsed.toolCalls[0]).toBe('run_shell_command');
});
test('extracts total tokens from result stats', () => {
const parsed = parseGeminiJSONL(FIXTURE_LINES);
expect(parsed.tokens).toBe(27147);
});
test('skips malformed lines without throwing', () => {
const lines = [
'{"type":"init","session_id":"ok"}',
'this is not json',
'{"type":"message","role":"assistant","content":"hello","delta":true}',
'{incomplete json',
'{"type":"result","status":"success","stats":{"total_tokens":100}}',
];
const parsed = parseGeminiJSONL(lines);
expect(parsed.sessionId).toBe('ok');
expect(parsed.output).toBe('hello');
expect(parsed.tokens).toBe(100);
});
test('skips empty and whitespace-only lines', () => {
const lines = [
'',
' ',
'{"type":"init","session_id":"s1"}',
'\t',
'{"type":"result","status":"success","stats":{"total_tokens":50}}',
];
const parsed = parseGeminiJSONL(lines);
expect(parsed.sessionId).toBe('s1');
expect(parsed.tokens).toBe(50);
});
test('handles empty input', () => {
const parsed = parseGeminiJSONL([]);
expect(parsed.output).toBe('');
expect(parsed.toolCalls).toHaveLength(0);
expect(parsed.tokens).toBe(0);
expect(parsed.sessionId).toBeNull();
});
test('handles missing fields gracefully', () => {
const lines = [
'{"type":"init"}', // no session_id
'{"type":"message","role":"assistant"}', // no content
'{"type":"tool_use"}', // no tool_name
'{"type":"result","status":"success"}', // no stats
];
const parsed = parseGeminiJSONL(lines);
expect(parsed.sessionId).toBeNull();
expect(parsed.output).toBe('');
expect(parsed.toolCalls).toHaveLength(0);
expect(parsed.tokens).toBe(0);
});
test('handles multiple tool_use events', () => {
const lines = [
'{"type":"tool_use","tool_name":"run_shell_command","tool_id":"cmd_1","parameters":{"command":"ls"}}',
'{"type":"tool_use","tool_name":"read_file","tool_id":"cmd_2","parameters":{"path":"foo.ts"}}',
'{"type":"tool_use","tool_name":"run_shell_command","tool_id":"cmd_3","parameters":{"command":"cat bar.ts"}}',
];
const parsed = parseGeminiJSONL(lines);
expect(parsed.toolCalls).toEqual(['run_shell_command', 'read_file', 'run_shell_command']);
});
});
-262
View File
@@ -1,262 +0,0 @@
/**
* Gemini CLI subprocess runner for skill E2E testing.
*
* Spawns `gemini -p` as an independent process, parses its stream-json
* output, and returns structured results. Follows the same pattern as
* codex-session-runner.ts but adapted for the Gemini CLI.
*
* Key differences from Codex session-runner:
* - Uses `gemini -p` instead of `codex exec`
* - Output is NDJSON with event types: init, message, tool_use, tool_result, result
* - Uses `--output-format stream-json --yolo` instead of `--json -s read-only`
* (`--skip-trust` was removed in gemini-cli 0.34; folder trust is settings-driven now)
* - No temp HOME needed — Gemini discovers skills from `.agents/skills/` in cwd
* - Message events are streamed with `delta: true` — must concatenate
*/
import * as path from 'path';
import { spawn } from 'child_process';
import { Readable } from 'node:stream';
import { hermeticChildEnv } from './hermetic-env';
import { killProcessGroup } from '../../scripts/test-strict-output';
// --- Interfaces ---
export interface GeminiResult {
output: string; // Full assistant message text (concatenated deltas)
toolCalls: string[]; // Tool names from tool_use events
tokens: number; // Total tokens used
exitCode: number; // Process exit code
durationMs: number; // Wall clock time
sessionId: string | null; // Session ID from init event
rawLines: string[]; // Raw JSONL lines for debugging
}
// --- JSONL parser ---
export interface ParsedGeminiJSONL {
output: string;
toolCalls: string[];
tokens: number;
sessionId: string | null;
}
/**
* Parse an array of JSONL lines from `gemini -p --output-format stream-json`.
* Pure function — no I/O, no side effects.
*
* Handles these Gemini event types:
* - init → extract session_id
* - message (role=assistant, delta=true) → concatenate content into output
* - tool_use → extract tool_name
* - tool_result → logged but not extracted
* - result → extract token usage from stats
*/
export function parseGeminiJSONL(lines: string[]): ParsedGeminiJSONL {
const outputParts: string[] = [];
const toolCalls: string[] = [];
let tokens = 0;
let sessionId: string | null = null;
for (const line of lines) {
if (!line.trim()) continue;
try {
const obj = JSON.parse(line);
const t = obj.type || '';
if (t === 'init') {
const sid = obj.session_id || '';
if (sid) sessionId = sid;
} else if (t === 'message') {
if (obj.role === 'assistant' && obj.content) {
outputParts.push(obj.content);
}
} else if (t === 'tool_use') {
const name = obj.tool_name || '';
if (name) toolCalls.push(name);
} else if (t === 'result') {
const stats = obj.stats || {};
tokens = (stats.total_tokens || 0);
}
} catch { /* skip malformed lines */ }
}
return {
output: outputParts.join(''),
toolCalls,
tokens,
sessionId,
};
}
// --- Main runner ---
/**
* Run a prompt via `gemini -p` and return structured results.
*
* Spawns gemini with stream-json output, parses JSONL events,
* and returns a GeminiResult. Skips gracefully if gemini binary is not found.
*/
export async function runGeminiSkill(opts: {
prompt: string; // What to ask Gemini
timeoutMs?: number; // Default 300000 (5 min)
cwd?: string; // Working directory (where .agents/skills/ lives)
}): Promise<GeminiResult> {
const {
prompt,
timeoutMs = 300_000,
cwd,
} = opts;
const startTime = Date.now();
// Check if gemini binary exists
const whichResult = Bun.spawnSync(['which', 'gemini'], { timeout: 30_000 });
if (whichResult.exitCode !== 0) {
return {
output: 'SKIP: gemini binary not found',
toolCalls: [],
tokens: 0,
exitCode: -1,
durationMs: Date.now() - startTime,
sessionId: null,
rawLines: [],
};
}
// Build gemini command.
// --skip-trust was REMOVED in gemini-cli 0.34 ("Unknown arguments:
// skip-trust"); folder trust moved to settings and no longer needs a flag
// for headless runs. --yolo still auto-approves tool actions.
const args = ['-p', prompt, '--output-format', 'stream-json', '--yolo'];
// Spawn gemini — uses real HOME for auth (~/.gemini; HOME is allowlisted),
// cwd for skill discovery. Hermetic scrub with gemini's auth surface
// re-admitted (previously this spawn inherited the full operator env).
// node:child_process spawn with `detached` (own process group) — mirrors
// session-runner.ts. A bare kill signalled only gemini itself; tool
// subprocesses survived as orphans holding our pipes open (the same
// blocked-drain hang the claude runner fixed — this copy lacked it).
const proc = spawn('gemini', args, {
cwd: cwd || process.cwd(),
stdio: ['ignore', 'pipe', 'pipe'],
detached: process.platform !== 'win32',
env: hermeticChildEnv(undefined, {
extraAllow: ['GEMINI_API_KEY', 'GOOGLE_API_KEY', 'GOOGLE_APPLICATION_CREDENTIALS', 'GOOGLE_CLOUD_*', 'GEMINI_*'],
}),
});
const stdoutWeb = Readable.toWeb(proc.stdout!) as ReadableStream<Uint8Array>;
const stderrWeb = Readable.toWeb(proc.stderr!) as ReadableStream<Uint8Array>;
const procExited: Promise<number> = new Promise((resolve) => {
proc.on('close', (code) => resolve(code ?? 1));
proc.on('error', () => resolve(1));
});
// Race against timeout
let timedOut = false;
const timeoutId = setTimeout(() => {
timedOut = true;
// Group SIGKILL + reader cancel: kill the whole tree AND unblock the
// read loop even if a stray grandchild survives the group kill.
killProcessGroup(proc, 'SIGKILL');
reader.cancel().catch(() => { /* stream already closed */ });
}, timeoutMs);
// Stream and collect JSONL from stdout
const collectedLines: string[] = [];
const stderrPromise = new Response(stderrWeb).text();
const reader = stdoutWeb.getReader();
const decoder = new TextDecoder();
let buf = '';
try {
while (true) {
const { done, value } = await reader.read();
if (done) break;
buf += decoder.decode(value, { stream: true });
const lines = buf.split('\n');
buf = lines.pop() || '';
for (const line of lines) {
if (!line.trim()) continue;
collectedLines.push(line);
// Real-time progress to stderr
try {
const event = JSON.parse(line);
if (event.type === 'tool_use' && event.tool_name) {
const elapsed = Math.round((Date.now() - startTime) / 1000);
process.stderr.write(` [gemini ${elapsed}s] tool: ${event.tool_name}\n`);
} else if (event.type === 'message' && event.role === 'assistant' && event.content) {
const elapsed = Math.round((Date.now() - startTime) / 1000);
process.stderr.write(` [gemini ${elapsed}s] message: ${event.content.slice(0, 100)}\n`);
}
} catch { /* skip — parseGeminiJSONL will handle it later */ }
}
}
} catch { /* stream read error — fall through to exit code handling */ }
// Flush remaining buffer
if (buf.trim()) {
collectedLines.push(buf);
}
// Same orphan hazard as stdout: a grandchild holding stderr open would
// block this drain forever. Race against child exit + a short grace window
// (ported from session-runner.ts — the gemini copy lacked it).
const stderr = await Promise.race([
stderrPromise,
(async () => {
await procExited;
await new Promise((r) => setTimeout(r, 5_000));
return '';
})(),
]);
const exitCode = await procExited;
clearTimeout(timeoutId);
const durationMs = Date.now() - startTime;
// Parse all collected JSONL lines
const parsed = parseGeminiJSONL(collectedLines);
// Log stderr if non-empty (may contain auth errors, etc.)
if (stderr.trim()) {
process.stderr.write(` [gemini stderr] ${stderr.trim().slice(0, 200)}\n`);
}
// Environment-unusable classification: these are Google-side conditions no
// test assertion can act on — the deprecated individual code-assist auth
// path ("migrate to the Antigravity suite") and argv drift on older/newer
// CLIs. Return the same SKIP shape as binary-not-found so callers report
// SKIPPED instead of a false FAIL.
const unusableMarkers = [
'no longer supported for Gemini Code Assist',
'antigravity',
'Unknown arguments: skip-trust',
];
if (exitCode !== 0 && parsed.tokens === 0) {
const marker = unusableMarkers.find((m) => stderr.toLowerCase().includes(m.toLowerCase()));
if (marker) {
return {
output: `SKIP: gemini CLI unusable (${marker})`,
toolCalls: [],
tokens: 0,
exitCode: -1,
durationMs,
sessionId: null,
rawLines: collectedLines,
};
}
}
return {
output: parsed.output,
toolCalls: parsed.toolCalls,
tokens: parsed.tokens,
exitCode: timedOut ? 124 : exitCode,
durationMs,
sessionId: parsed.sessionId,
rawLines: collectedLines,
};
}
-2
View File
@@ -21,6 +21,4 @@ export const OVERLAY_CASE_FILES: Record<string, string> = {
'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial.test.ts': 'opus-4-7-effort-match-trivial',
'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation.test.ts': 'opus-4-7-literal-interpretation',
'test/skill-e2e-overlay-harness-claude-dedicated-tools-vs-bash-sonnet.test.ts': 'claude-dedicated-tools-vs-bash-sonnet',
'test/skill-e2e-overlay-harness-opus-4-7-effort-match-trivial-sonnet.test.ts': 'opus-4-7-effort-match-trivial-sonnet',
'test/skill-e2e-overlay-harness-opus-4-7-literal-interpretation-sonnet.test.ts': 'opus-4-7-literal-interpretation-sonnet',
};
+2 -3
View File
@@ -11,8 +11,8 @@ import { matchGlob } from './touchfiles';
/** The exact globs package.json's `test:gate` passes to `bun test`. */
export const PAID_TEST_GLOBS = [
// skill-llm-eval* (not just the base file): skill-llm-eval-spec.test.ts
// fell outside the exact glob and could never run in any lane.
// skill-llm-eval* (not just the base file): a sibling judge file outside
// the exact glob could never run in any lane.
'test/skill-llm-eval*.test.ts',
'test/skill-e2e-*.test.ts',
'test/skill-routing-e2e.test.ts',
@@ -22,7 +22,6 @@ export const PAID_TEST_GLOBS = [
// entered the paid census. The same bug class as the deleted pre-split
// monolith (see test/paid-shards.test.ts's regression pin).
'test/codex-e2e*.test.ts',
'test/gemini-e2e.test.ts',
'test/llm-judge-recommendation.test.ts',
'test/carve-section-loading*.test.ts',
] as const;
+25 -13
View File
@@ -15,20 +15,32 @@
* the re-entry mechanism.
*/
export const PERIODIC_CI_EXCLUDE: Record<string, { reason: string; tracking: string }> = {
'test/skill-e2e-ship-idempotency.test.ts': {
reason:
'documented-red: the PTY child sits at the Claude Code welcome screen for the full budget '
+ '(readiness/typing race vs CLI 2.1.x); never green since it was born in v1.63',
tracking: 'TODOS.md "periodic tier — three documented-red tests need structural repair" (1 of 3 resolved: sidebar trio already deleted)',
'test/codex-e2e.test.ts': {
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/skill-e2e-brain-privacy-gate.test.ts': {
reason:
'documented-red: the artifacts-sync stop-gate preconditions do not survive the hermetic env '
+ 'even with per-test HOME/GSTACK_HOME injection; never green anywhere',
tracking: 'TODOS.md "periodic tier — three documented-red tests need structural repair"',
'test/codex-e2e-sol-scope.test.ts': {
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/skill-e2e-ios.test.ts': {
reason: 'requires a live iOS device/simulator toolchain (xcodebuild, devicectl) — manual hardware, not a CI runner capability',
tracking: 'TODOS.md "skill-e2e-ios CI story" (device/runner decision)',
'test/codex-e2e-shared-libs.test.ts': {
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/codex-e2e-recommendation-substance.test.ts': {
reason: 'the codex CLI is not installed in the CI image (Dockerfile.ci ships only claude-code); every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/skill-e2e-outside-voice.test.ts': {
reason: 'needs both the claude and codex CLIs; codex is not in the CI image, so every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/skill-e2e-aside.test.ts': {
reason: 'needs macOS with the Aside app open (asideAvailable()); CI runners are Linux, so every case self-skips',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
'test/skill-e2e-ios-device.test.ts': {
reason: 'needs a physical iPhone over USB/devicectl — manual hardware, not a CI runner capability',
tracking: 'TODOS.md "CI-unrunnable paid evals" (re-entry: the CLI/device is available in the CI image; review by 2026-12-28)',
},
};
-31
View File
@@ -13,37 +13,6 @@ interface PlanCountSnapshot {
claudeConfigDir: string | null;
}
/** Copy the caller-owned plan and review-log rows into an attempt's artifact
* directory before fixture cleanup removes them. Best-effort: each result
* (copied, missing or its error) is recorded in evidence-copy.json and never
* replaces the observation or its outcome. */
export function copyPlanCountEvidence(artifactDir: string | undefined,
evidence: { planPath?: string; reviewLogDirectory?: string }): void {
if (!artifactDir) return;
const results: Record<string, string> = {};
const copy = (label: string, source: string, target: string) => {
try {
if (!fs.existsSync(source)) { results[label] = `missing: ${source}`; return; }
fs.mkdirSync(path.dirname(target), { recursive: true, mode: 0o700 });
fs.copyFileSync(source, target);
fs.chmodSync(target, 0o600);
results[label] = `copied: ${source}`;
} catch (error) { results[label] = `error: ${String(error)}`; }
};
if (evidence.planPath) copy('plan', evidence.planPath, path.join(artifactDir, 'evidence', 'plan', path.basename(evidence.planPath)));
else results.plan = 'missing: no expected plan path';
if (evidence.reviewLogDirectory) {
let names: string[] = [];
try { names = fs.readdirSync(evidence.reviewLogDirectory).filter(name => name.endsWith('-reviews.jsonl')); }
catch (error) { results.reviewLog = fs.existsSync(evidence.reviewLogDirectory) ? `error: ${String(error)}` : `missing: ${evidence.reviewLogDirectory}`; }
if (!names.length && !results.reviewLog) results.reviewLog = `missing: no *-reviews.jsonl in ${evidence.reviewLogDirectory}`;
for (const name of names) copy(`reviewLog:${name}`, path.join(evidence.reviewLogDirectory, name), path.join(artifactDir, 'evidence', 'review-log', name));
} else results.reviewLog = 'missing: no fixture-owned review log binding';
try {
fs.writeFileSync(path.join(artifactDir, 'evidence-copy.json'), JSON.stringify(results, null, 2) + '\n', { mode: 0o600 });
} catch { /* the observation and its outcome stay authoritative */ }
}
/** One owned directory per count attempt; periodic captures replace files atomically. */
export function createPlanCountSnapshotWriter(env: NodeJS.ProcessEnv = process.env):
(input: PlanCountSnapshot) => { artifactDir?: string; artifactError?: string } {
+1 -11
View File
@@ -1,4 +1,4 @@
import { execFileSync, spawnSync } from 'node:child_process';
import { spawnSync } from 'node:child_process';
import * as fs from 'node:fs';
import * as path from 'node:path';
import type { NativeQuestion } from './plan-skill-questions';
@@ -54,16 +54,6 @@ export const DESIGN_BOARD_ACTOR_PROTOCOL = [
'decisions, visual verification, and the remaining review still apply.',
'',
].join('\n');
/** Declare the fixture user's interface before launch, without changing its plan. */
export function seedDesignBoardActorProtocol(cwd: string): void {
fs.appendFileSync(path.join(cwd, 'CLAUDE.md'), `\n${DESIGN_BOARD_ACTOR_PROTOCOL}`);
const git = (args: string[]) => execFileSync('git', args, { cwd, stdio: 'pipe', timeout: 10_000 });
git(['add', 'CLAUDE.md']);
git(['-c', 'user.name=Finding fixture', '-c', 'user.email=fixture@gstack.test', 'commit', '-m', 'Declare Design board fixture actor interface']);
git(['update-ref', 'refs/remotes/origin/main', 'HEAD']);
}
// The native picker is synchronous. Keep identity checks and the real board
// submission together in a bounded child; the counting driver still owns the
// subsequent terminal input and requires its native acknowledgment.
-27
View File
@@ -1,14 +1,5 @@
/** Ground truth for the existing plan fixtures, independent of review phase. */
import type { NativeQuestion } from './plan-skill-questions';
export const CEO_FINDINGS = [
{ id: 'dispatcher', description: 'Decide whether PaymentService should reuse the existing WebhookDispatcher instead of bypassing it for namespace separation.' },
{ id: 'sql', description: 'Decide a safe parameterized user lookup instead of interpolating untrusted request.params.userId into raw SQL.' },
{ id: 'email', description: 'Decide the failure/recovery contract for notification email after the payment transaction commits; the proposed inline email leg has no catch, outbox, or retry.' },
{ id: 'tests', description: 'Decide regression coverage for the new PaymentService path; existing platform tests do not cover this new handler and no new tests are planned.' },
{ id: 'queries', description: 'Decide how to eliminate or justify the per-order lookup loop instead of batching the webhook order reads.' },
];
export const CEO_PAIRED_FINDINGS = [
{ id: 'receipt-test', description: 'Independently decide happy-path processPayment test coverage asserting the correct receipt after a successful Stripe charge.' },
{ id: 'failure-test', description: 'Independently decide Stripe 502/timeout coverage asserting one retry with backoff and then clean failure.' },
@@ -21,15 +12,6 @@ export const CEO_SCOPE_CANDIDATES = [
{ id: 'E4', description: 'The whole Telegram bot API integration: include, defer, or cut it.' },
{ id: 'E5', description: 'The whole Mattermost REST plugin integration: include, defer, or cut it.' },
];
export const DESIGN_FINDINGS = [
{ id: 'primary-action', description: 'Decide how Save becomes the primary action instead of sharing equal visual weight with Reset, Cancel, and Export.' },
{ id: 'spacing', description: 'Decide a consistent vertical section rhythm instead of mixed 16px, 24px, and 32px gaps.' },
{ id: 'contrast', description: 'Decide accessible error-message contrast instead of the proposed approximately 3:1 red-on-pink treatment.' },
{ id: 'typography', description: 'Decide a coherent form-label type hierarchy instead of inconsistent 14px, 16px, and 18px labels.' },
{ id: 'save-feedback', description: 'Decide progress feedback for the 2–5 second Save action instead of leaving the interface apparently frozen.' },
];
export const DEVEX_FINDINGS = [
{ id: 'persona', description: 'Decide a specific target developer persona instead of shipping for everyone.' },
{ id: 'first-run-benchmark', description: 'Decide measurement/benchmarking of time to hello world instead of leaving first-run duration unknown.' },
@@ -37,15 +19,6 @@ export const DEVEX_FINDINGS = [
{ id: 'aha', description: 'Decide a concrete interactive demo or aha moment in the getting-started flow instead of documentation alone.' },
{ id: 'peer-comparison', description: 'Produce grounded comparative analysis of peer SDK developer experiences and its implications for this plan instead of ignoring existing solutions.' },
];
export const ENG_FINDINGS = [
{ id: 'shared-cache', description: 'Decide ownership/isolation of AuthCache instead of two services mutating one module-level global cache.' },
{ id: 'swallowed-errors', description: 'Decide explicit handling of the swallowed error classes in validateAndDispatch rather than keeping three nested catch blocks that hide failures.' },
{ id: 'legacy-regression', description: 'Decide regression tests for the rewritten legacyAuthFlow behavior.' },
{ id: 'parallel-idp', description: 'Decide parallelizing the five independent IDP validation calls instead of running them sequentially.' },
{ id: 'complexity', description: 'Decide whether to reduce or justify the 12-file/four-new-class scope and complexity.' },
];
export const ENG_BATCHING_FINDINGS = [
{ id: 'retry-library', description: 'Decide reuse of existing job-library retry hooks instead of a custom inline backoff scheduler per worker.' },
{ id: 'retry-duplication', description: 'Decide consolidation of the copied retry envelope across five workers.' },
-81
View File
@@ -1,81 +0,0 @@
import { readOwnedClaudeTranscript } from './owned-claude-transcript';
/** A report preview is not completion. Require the latest owned conversation
* turn to finish without tools, then corroborate its own completion text.
*/
export function readPlanSkillCompletion(configDir: string | null, sessionId: string, visible: string): string | null {
const transcript = readOwnedClaudeTranscript(configDir, sessionId);
if (transcript.pendingBytes) return null;
const queued = new Map<string | undefined, number>();
let queuedCount = 0;
let ambiguousDequeues = false;
let latest: { id: string | null; text: string[]; stop: unknown; tools: boolean } | null = null;
for (const row of transcript.rows) {
if (row.type === 'queue-operation') {
latest = null;
const hasContent = Object.hasOwn(row, 'content');
if (!['enqueue', 'dequeue', 'remove', 'popOne', 'popAll'].includes(row.operation)
|| hasContent && (typeof row.content !== 'string' || row.operation === 'dequeue')) {
throw new Error('Unsupported queue operation in owned Claude transcript');
}
const count = queued.get(row.content) ?? 0;
if (row.operation === 'enqueue') {
if (ambiguousDequeues) throw new Error('Unsupported mixed queue history after anonymous dequeue in owned Claude transcript');
queued.set(row.content, count + 1);
queuedCount++;
} else {
// Claude 2.1.263 emits one removal per actual item, even for popAll.
// Dequeue omits identity: payload counts remain upper bounds until
// the logged queue drains. New enqueues during unresolved ambiguity
// fail closed; this reader cannot reconstruct every producer stream.
if (!queuedCount) throw new Error('Queue removal lacks its enqueue in owned Claude transcript');
if (row.operation === 'dequeue') ambiguousDequeues = queued.size > 1;
else {
if (!count) throw new Error('Queue removal does not match a queued payload in owned Claude transcript');
if (count > 1) queued.set(row.content, count - 1);
else queued.delete(row.content);
}
if (--queuedCount === 0) { queued.clear(); ambiguousDequeues = false; }
else if (queued.size === 1) {
// The remaining payload is now unique, even if it had duplicates.
queued.set(queued.keys().next().value, queuedCount);
ambiguousDequeues = false;
}
}
continue;
}
if (row.type === 'user' || (row.type === 'attachment' && typeof row.attachment?.prompt === 'string')) {
latest = null;
continue;
}
if (row.type !== 'assistant' || row.message?.role !== 'assistant') continue;
const message = row.message;
const id = typeof message.id === 'string' ? message.id : null;
if (!latest || !id || latest.id !== id) latest = { id, text: [], stop: null, tools: false };
latest.stop = message.stop_reason;
if (!Array.isArray(message.content)) continue;
for (const block of message.content) {
if (block?.type === 'text' && typeof block.text === 'string') latest.text.push(block.text);
if (block?.type === 'tool_use') latest.tools = true;
}
}
if (queuedCount || !latest || latest.stop !== 'end_turn' || latest.tools) return null;
let fence: string | null = null;
const compact = (value: string) => value.replace(/[\s*#]/g, '').toLowerCase();
for (const line of latest.text.join('\n').split(/\r?\n/)) {
const delimiter = line.match(/^ {0,3}(`{3,}|~{3,})/)?.[1];
if (delimiter) {
if (!fence) fence = delimiter;
else if (delimiter[0] === fence[0] && delimiter.length >= fence.length) fence = null;
continue;
}
if (fence || /^(?: {4}| {0,3}\t)|^\s*>/.test(line)) continue;
const text = line.replace(/^ {0,3}(?:#{1,6}\s+)?/, '').replace(/\*\*/g, '').trim();
const marker = /^(?:GSTACK REVIEW REPORT|Completion Summary)$/i.test(text)
|| /^VERDICT:\s*\S/.test(text)
|| /^Status:\s*(?:clean|issues_open)\b/i.test(text)
|| /^(?:STATUS:\s*)?DONE(?:_WITH_CONCERNS)?(?:\s|[—:.-]|$)/.test(text);
if (marker && compact(visible).includes(compact(text))) return text;
}
return null;
}
+2 -2
View File
@@ -3,8 +3,8 @@ import * as path from 'node:path';
/**
* Provider adapter interface — uniform contract for Claude, GPT, Gemini.
*
* Each adapter wraps an existing runner (session-runner.ts, codex-session-runner.ts,
* gemini-session-runner.ts) and normalizes its per-provider result shape into the
* Each adapter wraps an existing runner or CLI (session-runner.ts,
* codex-session-runner.ts, the gemini CLI) and normalizes its per-provider result shape into the
* RunResult below. The benchmark harness only talks to adapters through this
* interface, never to the underlying runners directly.
*/
-148
View File
@@ -1,148 +0,0 @@
import type { Terminal } from 'xterm';
let TerminalClass: typeof Terminal | undefined;
function terminalClass(): typeof Terminal {
if (TerminalClass) return TerminalClass;
// xterm 5.3 detects Node by navigator's absence. Bun provides navigator
// without a DOM, so select the package's Node path during synchronous load.
// No await or fake document/window: restore the exact descriptors even if
// loading fails. Its public buffer API works without Terminal.open().
const navigator = Object.getOwnPropertyDescriptor(globalThis, 'navigator');
const self = Object.getOwnPropertyDescriptor(globalThis, 'self');
try {
Object.defineProperty(globalThis, 'navigator', { configurable: true, value: undefined });
if (typeof globalThis.self === 'undefined') Object.defineProperty(globalThis, 'self', { configurable: true, value: globalThis });
TerminalClass = (require('xterm') as { Terminal: typeof Terminal }).Terminal;
return TerminalClass;
} finally {
if (navigator) Object.defineProperty(globalThis, 'navigator', navigator);
else delete (globalThis as any).navigator;
if (self) Object.defineProperty(globalThis, 'self', self);
else delete (globalThis as any).self;
}
}
export interface PtyScreenSnapshot {
/** Absolute bytes fed through this snapshot's write barrier, not raw JS string indices. */
inputOffset: number;
cols: number;
rows: number;
bufferType: 'normal' | 'alternate';
lines: Array<{ text: string; wrapped: boolean }>;
text: string;
/** Contiguous dim/inverse text cells, captured at the same write barrier. */
styledText: Array<{ row: number; start: number; text: string; dim: boolean; inverse: boolean }>;
}
/** Test-only current-screen projection. Never writes input to a PTY. */
export class PtyCurrentScreen {
private terminal: Terminal | null = null;
private offset = 0;
private closed: Error | null = null;
private pending = new Set<(error: Error) => void>();
private readonly cols: number;
private readonly rows: number;
private readonly flushTimeoutMs: number;
constructor(options: { cols?: number; rows?: number; flushTimeoutMs?: number } = {}) {
this.cols = options.cols ?? 120;
this.rows = options.rows ?? 40;
this.flushTimeoutMs = options.flushTimeoutMs ?? 1000;
if (![this.cols, this.rows].every(value => Number.isSafeInteger(value) && value > 0)) throw new Error('Screen dimensions must be positive integers');
if (!Number.isFinite(this.flushTimeoutMs) || this.flushTimeoutMs <= 0) throw new Error('Screen flush timeout must be finite and positive');
}
get inputOffset(): number { return this.offset; }
private getTerminal(): Terminal {
if (this.closed) throw this.closed;
if (!this.terminal) this.terminal = new (terminalClass())({ cols: this.cols, rows: this.rows, scrollback: 0 });
return this.terminal;
}
/** Byte chunks preserve decoder state across split UTF-8 characters.
* Strings are encoded once; offsets always count UTF-8 bytes. */
feed(data: string | Uint8Array): number {
const terminal = this.getTerminal();
const bytes = typeof data === 'string' ? new TextEncoder().encode(data) : new Uint8Array(data);
const offset = this.offset + bytes.byteLength;
if (!Number.isSafeInteger(offset)) throw new Error('Screen input offset exceeded the safe integer range');
try { terminal.write(bytes); }
catch (cause) {
this.close(new Error('Screen input could not be decoded'));
throw cause;
}
this.offset = offset;
return offset;
}
/** Capture inside the callback, before later queued writes can change the
* screen. The returned offset lets callers compare with their input epoch;
* it does not itself establish native question ownership or acknowledgement. */
snapshot(): Promise<PtyScreenSnapshot> {
const terminal = this.getTerminal();
const inputOffset = this.offset;
return new Promise((resolve, reject) => {
let done = false;
const finish = (error?: Error, value?: PtyScreenSnapshot) => {
if (done) return;
done = true;
clearTimeout(timer);
this.pending.delete(fail);
if (error) reject(error);
else resolve(value!);
};
const fail = (error: Error) => finish(error);
const timer = setTimeout(() => this.close(new Error(`Screen flush did not complete within ${this.flushTimeoutMs}ms`)), this.flushTimeoutMs);
this.pending.add(fail);
try {
terminal.write('', () => {
if (done || this.closed) return;
try {
const buffer = terminal.buffer.active;
const styledText: PtyScreenSnapshot['styledText'] = [];
const lines = Array.from({ length: terminal.rows }, (_, row) => {
const line = buffer.getLine(buffer.baseY + row);
let start = 0, previous = '';
for (let col = 0; col <= terminal.cols; col++) {
const cell = col < terminal.cols ? line?.getCell(col) : undefined;
const key = cell && (cell.getChars() || cell.getWidth() === 0)
? `${Number(!!cell.isDim())}${Number(!!cell.isInverse())}` : '';
if (key === previous) continue;
if (previous && previous !== '00') styledText.push({ row, start,
text: line!.translateToString(false, start, col), dim: previous[0] === '1', inverse: previous[1] === '1' });
start = col; previous = key;
}
return { text: line?.translateToString(true) ?? '', wrapped: line?.isWrapped ?? false };
});
finish(undefined, { inputOffset, cols: terminal.cols, rows: terminal.rows,
bufferType: buffer.type, lines, text: lines.map(line => line.text).join('\n'), styledText });
} catch (cause) {
this.close(cause instanceof Error ? cause : new Error(String(cause)));
}
});
} catch (cause) {
this.close(cause instanceof Error ? cause : new Error(String(cause)));
}
});
}
/** The caller must flush before coordinating this with the actual PTY.
* Resizing changes geometry only; it contributes no output or input epoch. */
resize(cols: number, rows: number): void {
if (![cols, rows].every(value => Number.isSafeInteger(value) && value > 0)) throw new Error('Screen dimensions must be positive integers');
if (this.pending.size) throw new Error('Cannot resize while a screen snapshot is pending');
try { this.getTerminal().resize(cols, rows); }
catch (cause) { this.close(cause instanceof Error ? cause : new Error(String(cause))); throw cause; }
}
private close(error: Error): void {
if (this.closed) return;
this.closed = error;
for (const fail of [...this.pending]) fail(error);
this.terminal?.dispose();
}
dispose(): void { this.close(new Error('Screen projection is disposed')); }
}
-40
View File
@@ -1,40 +0,0 @@
/**
* requiredReads enforcement (v2 plan T9, mitigation layer 5 — the only CI-failing
* layer against silent section-skip).
*
* Given a /ship run's tool calls and the set of section files the run's SITUATION
* required, assert the agent actually Read each one. The required set comes from
* the TEST FIXTURE (which situation it set up), NOT from the manifest — the
* manifest is passive (CM2). This keeps "when is a section required" in exactly
* one machine-checkable place: the eval fixtures.
*
* Builds on extractSectionReads from transcript-section-logger so section-path
* matching (the `/sections/<file>.md` segment, host-layout agnostic) lives in one
* place.
*/
import { extractSectionReads, type TranscriptResultLike } from './transcript-section-logger';
export interface RequiredReadsResult {
required: string[];
read: string[];
missing: string[];
ok: boolean;
}
/**
* @param result the skill run (anything with toolCalls)
* @param requiredFiles section basenames the situation required, e.g.
* ['version-bump.md','changelog.md'] (or with a sections/
* prefix — normalized to basename here)
*/
export function assertRequiredReads(
result: TranscriptResultLike,
requiredFiles: string[],
): RequiredReadsResult {
const read = extractSectionReads(result);
const readSet = new Set(read);
const required = requiredFiles.map(f => f.replace(/^.*\//, '')); // tolerate sections/<f>
const missing = required.filter(f => !readSet.has(f));
return { required, read, missing, ok: missing.length === 0 };
}
+44
View File
@@ -0,0 +1,44 @@
import { describe, expect, test } from 'bun:test';
import * as path from 'path';
import { directSpecifiers, resolveRepoLiteral, resolveRepoSpecifier } from './resolve-repo-path';
const ROOT = path.resolve(import.meta.dir, '../..');
describe('resolve-repo-path', () => {
test('relative specifiers resolve to repo files, with and without extensions', () => {
expect(resolveRepoSpecifier(ROOT, 'test/touchfiles.test.ts', './helpers/touchfiles')).toBe('test/helpers/touchfiles.ts');
expect(resolveRepoSpecifier(ROOT, 'test/touchfiles.test.ts', '../lib/eval-model')).toBe('lib/eval-model.ts');
expect(resolveRepoSpecifier(ROOT, 'lib/code-intelligence/gbrain-adapter.ts', '../egress-receipt.js')).toBe('lib/egress-receipt.ts');
expect(resolveRepoSpecifier(ROOT, 'test/touchfiles.test.ts', './helpers/no-such-module')).toBeNull();
});
test('bare packages and bun:/node: builtins never resolve', () => {
for (const specifier of ['bun:test', 'node:fs', 'fs', '@anthropic-ai/sdk', 'ts-morph']) {
expect(resolveRepoSpecifier(ROOT, 'test/touchfiles.test.ts', specifier)).toBeNull();
}
});
test('direct specifiers include imports, re-exports, require and literal dynamic import, not type-only imports', () => {
const source = [
"import { a } from './a';",
"import type { T } from './types';",
"export type { U } from './more-types';",
"export { b } from '../lib/b';",
"import {\n c,\n d,\n} from './multi';",
"import './side-effect';",
"const e = require('./e');",
"const f = await import('./f');",
"const g = await import(name);",
].join('\n');
expect(directSpecifiers(source)).toEqual(['./a', '../lib/b', './multi', './side-effect', './e', './f']);
});
test('path literals resolve only when the repo path exists', () => {
expect(resolveRepoLiteral(ROOT, 'bin/gstack-config')).toBe('bin/gstack-config');
expect(resolveRepoLiteral(ROOT, 'test/fixtures/')).toBe('test/fixtures');
expect(resolveRepoLiteral(ROOT, path.join('test', 'helpers', 'touchfiles.ts'))).toBe('test/helpers/touchfiles.ts');
expect(resolveRepoLiteral(ROOT, 'test/fixtures/no-such-fixture.json')).toBeNull();
expect(resolveRepoLiteral(ROOT, '/etc/passwd')).toBeNull();
expect(resolveRepoLiteral(ROOT, '../outside')).toBeNull();
});
});
+44
View File
@@ -0,0 +1,44 @@
/**
* Resolve a module specifier or path literal from a repo file to a repo-relative
* path, or null when it names no file in the checkout. Bare packages and
* `bun:` / `node:` builtins never resolve. Shared by the touchfile closure
* invariant and the test-of-test ratchet.
*/
import * as fs from 'fs';
import * as path from 'path';
const isFile = (candidate: string) => fs.existsSync(candidate) && fs.statSync(candidate).isFile();
/** A relative module specifier (`./x`, `../lib/y.js`) → repo-relative file, or null. */
export function resolveRepoSpecifier(root: string, fromFile: string, specifier: string): string | null {
if (!specifier.startsWith('.')) return null;
const base = path.resolve(path.dirname(path.join(root, fromFile)), specifier);
for (const candidate of [base, `${base}.ts`, base.replace(/\.js$/, '.ts'), `${base}.tsx`, path.join(base, 'index.ts')]) {
if (isFile(candidate)) return toRelative(root, candidate);
}
return null;
}
/** A repo-rooted path literal (`test/fixtures/x.json`, `bin/gstack-x`) → itself when it exists, else null. */
export function resolveRepoLiteral(root: string, literal: string): string | null {
const clean = literal.replace(/\/+$/, '');
if (!clean || path.isAbsolute(clean) || clean.startsWith('..')) return null;
return fs.existsSync(path.join(root, clean)) ? clean : null;
}
function toRelative(root: string, absolute: string): string | null {
const relative = path.relative(root, absolute).split(path.sep).join('/');
return relative.startsWith('..') ? null : relative;
}
/** Direct module specifiers of a source file: static imports/re-exports, require() and literal dynamic import(); `import type` excluded. */
export function directSpecifiers(source: string): string[] {
const out: string[] = [];
const typeOnly = /^\s*(?:import|export)\s+type\s/;
for (const match of source.matchAll(/^[ \t]*(?:import|export)\b[^;'"`]*?from\s*(['"])([^'"]+)\1|^[ \t]*import\s*(['"])([^'"]+)\3/gm)) {
if (typeOnly.test(match[0])) continue;
out.push(match[2] ?? match[4]!);
}
for (const match of source.matchAll(/\b(?:require|import)\s*\(\s*(['"])([^'"]+)\1\s*\)/g)) out.push(match[2]!);
return out;
}
-12
View File
@@ -33,18 +33,6 @@ export function gitArgvIn(repoDir: string, args: string[], timeout = 5000, env?:
return spawnSync('git', [...GIT_HERMETIC_ARGS, ...args], { cwd: repoDir, timeout, env });
}
/** Create a scratch repo (mkdtemp) with an initial commit; caller cleans up. */
export function makeScratchRepo(prefix: string, files: Record<string, string> = { 'src.txt': 'v1\n' }): string {
const repoDir = fs.mkdtempSync(path.join(os.tmpdir(), prefix));
gitIn(repoDir, 'init -q -b main');
for (const [name, content] of Object.entries(files)) {
fs.writeFileSync(path.join(repoDir, name), content);
}
gitIn(repoDir, `add ${Object.keys(files).join(' ')}`);
gitIn(repoDir, 'commit -q -m init');
return repoDir;
}
/** Recursively find files with a given suffix under a directory. */
export function findFilesBySuffix(root: string, suffix: string): string[] {
const found: string[] = [];
+63
View File
@@ -0,0 +1,63 @@
/**
* Derived touchfile rule: a paid test depends on every `test/helpers` and
* `test/fixtures` file it reaches through static imports, plus every such path
* named in a string literal inside that closure (fixtures read through `fs`).
* The result is a lower bound: a computed fixture path is declared by adding it
* to the key by hand.
*/
import * as fs from 'fs';
import * as path from 'path';
import { matchGlob } from './test-selection';
import { resolveRepoLiteral, resolveRepoSpecifier } from './resolve-repo-path';
export interface ClosureEntry {
/** Repo-relative dependency path. */
file: string;
/** How the paid test reached it: the import chain, ending in `"literal"` for string references. */
chain: string[];
}
const LITERAL = /test\/(?:helpers|fixtures)\/[A-Za-z0-9_.\-/]+[A-Za-z0-9_]/g;
const scanner = new Bun.Transpiler({ loader: 'tsx' });
/** Selection itself is diffed by map (diffTouchfileMaps), and test files are never dependencies. */
const SELECTION_MODULES = new Set(['test/helpers/touchfiles-data.ts', 'test/helpers/touchfiles.ts', 'test/helpers/test-selection.ts']);
const inScope = (file: string) => (file.startsWith('test/helpers/') || file.startsWith('test/fixtures/')) &&
!file.endsWith('.test.ts') && !SELECTION_MODULES.has(file);
/**
* Static test/helpers + test/fixtures closure of one paid test file, with literal fixture paths.
* Traversal stops at `boundary` files (the global touchfiles): an edit there already selects every test.
*/
export function paidTestClosure(root: string, testFile: string, boundary: ReadonlySet<string> = new Set()): ClosureEntry[] {
const seen = new Map<string, string[]>();
const queue: Array<{ file: string; chain: string[] }> = [{ file: testFile, chain: [testFile] }];
while (queue.length) {
const { file, chain } = queue.shift()!;
if (!/\.(ts|tsx|js|mjs)$/.test(file) || boundary.has(file)) continue;
const source = fs.readFileSync(path.join(root, file), 'utf8');
let imports: Array<{ path: string }> = [];
try { imports = scanner.scanImports(source); } catch { imports = []; }
for (const { path: specifier } of imports) {
const target = resolveRepoSpecifier(root, file, specifier);
if (!target || !inScope(target) || seen.has(target)) continue;
seen.set(target, [...chain, target]);
queue.push({ file: target, chain: [...chain, target] });
}
for (const match of source.match(LITERAL) ?? []) {
const found = inScope(match) ? resolveRepoLiteral(root, match) : null;
if (!found) continue;
const literal = fs.statSync(path.join(root, found)).isDirectory() ? `${found}/**` : found;
if (seen.has(literal)) continue;
seen.set(literal, [...chain, `"${match}"`]);
if (!literal.endsWith('/**')) queue.push({ file: literal, chain: [...chain, `"${match}"`] });
}
}
return [...seen].map(([file, chain]) => ({ file, chain }));
}
/** True when a dependency is selected by the key's own list or the global list. */
export function isCovered(file: string, patterns: readonly string[], globals: readonly string[]): boolean {
const probe = file.endsWith('/**') ? `${file.slice(0, -3)}/probe` : file;
return [...patterns, ...globals].some(pattern => pattern === file || matchGlob(probe, pattern));
}
File diff suppressed because it is too large. Load diff
-196
View File
@@ -1,196 +0,0 @@
/**
* Transcript section logger (v2 plan T10).
*
* Two jobs, both pure analysis over a SkillTestResult / NDJSON transcript:
*
* 1. extractSectionReads() — which `sections/*.md` files a run actually Read.
* Used by the sectioned world (post-carve) to verify the agent opened the
* chapters its situation required.
*
* 2. extractShipActions() — an observable ACTION fingerprint of a /ship run
* (ran tests, bumped VERSION, wrote CHANGELOG, created PR, ...). This works
* on BOTH the monolith and the sectioned skill, which is the whole point:
* capture a baseline on the current monolith ship FIRST, then assert the
* sectioned ship still performs the same actions. A section-read check alone
* can't catch "agent read the chapter but skipped the step"; the action
* fingerprint can.
*
* Why baseline-first (Codex outside-voice critique on the T9 plan): a logger
* shipped in the same PR as the carve is post-failure telemetry unless it has a
* pre-carve reference. captureShipBaseline() records the monolith's action
* fingerprint so compareShipActions() can flag a regression introduced by the
* carve.
*
* Pure functions, no I/O except the explicit read/write baseline helpers. The
* unit tests drive these with synthetic transcripts — no paid run needed to
* validate the logic.
*/
import * as fs from 'fs';
import * as path from 'path';
import * as os from 'os';
/** Minimal shape we need from SkillTestResult — kept structural so callers can
* pass a full SkillTestResult or a hand-built fixture in unit tests. */
export interface ToolCallLike {
tool: string;
input: unknown;
output?: string;
}
export interface TranscriptResultLike {
toolCalls: ToolCallLike[];
output?: string;
}
/** Pull the file_path off a tool-call input, tolerating unknown shapes. */
function readFilePath(input: unknown): string | null {
if (input && typeof input === 'object') {
const fp = (input as Record<string, unknown>).file_path;
if (typeof fp === 'string') return fp;
}
return null;
}
/** Pull the command string off a Bash tool-call input. */
function bashCommand(input: unknown): string | null {
if (input && typeof input === 'object') {
const cmd = (input as Record<string, unknown>).command;
if (typeof cmd === 'string') return cmd;
}
return null;
}
/**
* Every `sections/<name>.md` file the run Read, normalized to the section
* basename (e.g. "version-bump.md"). Deduped, in first-Read order. Matching is
* on the path segment `/sections/<file>.md` so it works regardless of whether
* the host resolved a relative, absolute, or prefixed install path.
*/
export function extractSectionReads(result: TranscriptResultLike): string[] {
const seen = new Set<string>();
const ordered: string[] = [];
for (const call of result.toolCalls) {
if (call.tool !== 'Read') continue;
const fp = readFilePath(call.input);
if (!fp) continue;
const m = fp.match(/(?:^|\/)sections\/([A-Za-z0-9._-]+\.md)$/);
if (!m) continue;
const name = m[1];
if (!seen.has(name)) {
seen.add(name);
ordered.push(name);
}
}
return ordered;
}
/**
* The canonical /ship action vocabulary. Each action is detected from the Bash
* commands the agent ran (plus a couple of Write/Edit signals). Order is the
* rough ship sequence; detection is order-independent.
*
* Keep this list aligned with the ship skeleton's numbered steps. The
* section-loading eval asserts the sectioned ship still triggers the same
* actions a monolith run did for the same fixture situation.
*/
export const SHIP_ACTIONS = [
'merged_base', // git merge <base>
'ran_tests', // bun test / npm test / the project test cmd
'bumped_version', // wrote VERSION / package.json version / ran gstack-version-bump
'wrote_changelog', // edited CHANGELOG.md
'committed', // git commit
'pushed', // git push
'opened_pr', // gh pr create / glab mr create
] as const;
export type ShipAction = (typeof SHIP_ACTIONS)[number];
const BASH_ACTION_PATTERNS: Array<{ action: ShipAction; re: RegExp }> = [
{ action: 'merged_base', re: /\bgit\s+merge\b/ },
{ action: 'ran_tests', re: /\b(bun\s+test|npm\s+(run\s+)?test|yarn\s+test|pytest|go\s+test|cargo\s+test|rspec)\b/ },
{ action: 'bumped_version', re: /gstack-version-bump\b|gstack-next-version\b|>\s*VERSION\b|npm\s+version\b/ },
{ action: 'wrote_changelog', re: /CHANGELOG\.md/ },
{ action: 'committed', re: /\bgit\s+commit\b/ },
{ action: 'pushed', re: /\bgit\s+push\b/ },
{ action: 'opened_pr', re: /\bgh\s+pr\s+create\b|\bglab\s+mr\s+create\b/ },
];
/**
* The observable action fingerprint of a ship run. Works on monolith AND
* sectioned skills because it reads what the agent DID (Bash + file writes),
* not which prose it loaded.
*/
export function extractShipActions(result: TranscriptResultLike): ShipAction[] {
const found = new Set<ShipAction>();
for (const call of result.toolCalls) {
if (call.tool === 'Bash') {
const cmd = bashCommand(call.input);
if (!cmd) continue;
for (const { action, re } of BASH_ACTION_PATTERNS) {
if (re.test(cmd)) found.add(action);
}
} else if (call.tool === 'Write' || call.tool === 'Edit') {
const fp = readFilePath(call.input);
if (fp && /CHANGELOG\.md$/.test(fp)) found.add('wrote_changelog');
if (fp && /(?:^|\/)VERSION$/.test(fp)) found.add('bumped_version');
}
}
// Preserve canonical order.
return SHIP_ACTIONS.filter(a => found.has(a));
}
export interface ShipBaseline {
tag: string;
/** Fixture/situation id this baseline was captured for. */
situation: string;
/** Action fingerprint observed on the monolith ship. */
actions: ShipAction[];
/** Section reads observed (empty on the monolith — present after carve). */
sectionReads: string[];
capturedAt: string;
}
const DEFAULT_BASELINE_DIR = path.join(os.homedir(), '.gstack-dev', 'ship-baselines');
/** Where a baseline for a given situation lives. */
export function baselinePath(situation: string, dir = DEFAULT_BASELINE_DIR): string {
return path.join(dir, `${situation}.json`);
}
/** Persist a ship baseline (used once on the monolith, before the carve). */
export function writeShipBaseline(baseline: ShipBaseline, dir = DEFAULT_BASELINE_DIR): string {
fs.mkdirSync(dir, { recursive: true });
const p = baselinePath(baseline.situation, dir);
fs.writeFileSync(p, JSON.stringify(baseline, null, 2) + '\n');
return p;
}
/** Read a previously-captured baseline, or null if none exists yet. */
export function readShipBaseline(situation: string, dir = DEFAULT_BASELINE_DIR): ShipBaseline | null {
try {
return JSON.parse(fs.readFileSync(baselinePath(situation, dir), 'utf-8')) as ShipBaseline;
} catch {
return null;
}
}
export interface ShipActionDiff {
/** Actions the baseline performed that the current run did NOT (the regression set). */
missing: ShipAction[];
/** Actions the current run performed that the baseline did not (usually fine). */
added: ShipAction[];
/** True when no baseline action was dropped. */
ok: boolean;
}
/**
* Compare a current sectioned-ship run against the monolith baseline. A dropped
* action (in baseline, not in current) is the carve regression we care about:
* the sectioned ship stopped doing something the monolith did.
*/
export function compareShipActions(baseline: ShipBaseline, current: ShipAction[]): ShipActionDiff {
const cur = new Set(current);
const base = new Set(baseline.actions);
const missing = baseline.actions.filter(a => !cur.has(a));
const added = current.filter(a => !base.has(a));
return { missing, added, ok: missing.length === 0 };
}